{
  "aliases": [
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-common_sense_qa_2_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-science_qa_external",
        "epoch-superglue_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.11990",
          "line": 10,
          "metricId": "Score",
          "observedAt": "2022-01-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.354",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=14;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.354
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.11990",
          "line": 12,
          "metricId": "Score",
          "observedAt": "2022-01-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.339",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=16;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.339
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.11990",
          "line": 14,
          "metricId": "Score",
          "observedAt": "2022-01-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.34",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=18;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.34
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 75
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "text-davinci-001",
      "numericRowCount": 75,
      "observedAtMax": "2022-01-27",
      "observedAtMin": "2022-01-27",
      "rowCount": 75,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 75
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 536,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.829,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=48;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.829
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1537,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.0628889999999993,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=48;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.0628889999999993
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4482,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=48;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Amazon Nova Lite",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 547,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.794,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=49;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.794
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1548,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 0.8952520000000004,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=49;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 0.8952520000000004
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4533,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=49;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Amazon Nova Micro",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 525,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.87,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=47;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.87
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1526,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.5656869999999996,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=47;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.5656869999999996
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4431,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=47;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Amazon Nova Pro",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 404,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.768,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=36;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.768
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1405,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.9610197002887726,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=36;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.9610197002887726
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3870,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=36;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Arctic Instruct",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 580,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.583,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=52;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.583
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1581,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.857238686800003,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=52;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.857238686800003
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4686,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=52;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 2.0",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 591,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.604,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=53;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.604
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1592,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 7.7061755385398865,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=53;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 7.7061755385398865
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4737,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=53;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 2.1",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 602,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.699,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=54;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.699
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1603,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.2278449382781982,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=54;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.2278449382781982
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4788,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=54;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 3 Haiku (20240307)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 624,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.924,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=56;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.924
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1625,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 7.469249876976013,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=56;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 7.469249876976013
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4890,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=56;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 3 Opus (20240229)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 613,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.907,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=55;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.907
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1614,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.2127642614841463,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=55;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.2127642614841463
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4839,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=55;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 3 Sonnet (20240229)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 635,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.815,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=57;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.815
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1636,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.915386771917343,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=57;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.915386771917343
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4941,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=57;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 3.5 Haiku (20241022)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 646,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.949,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=58;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.949
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1647,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.162740940093994,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=58;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.162740940093994
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4992,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=58;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 3.5 Sonnet (20240620)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 657,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.956,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=59;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.956
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1658,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.5175547733306884,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=59;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.5175547733306884
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5043,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=59;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude 3.5 Sonnet (20241022)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 569,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.721,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=51;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.721
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1570,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.474282945394516,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=51;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.474282945394516
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4635,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=51;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude Instant 1.2",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 558,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.784,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=50;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.784
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1559,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.653211696863174,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=50;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.653211696863174
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4584,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=50;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Claude v1.3",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 668,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.452,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=60;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.452
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1669,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.127378141641617,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=60;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.127378141641617
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5094,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=60;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Command",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 679,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.149,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=61;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.149
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1680,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.7514978868961335,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=61;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.7514978868961335
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5145,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=61;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Command Light",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 690,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.551,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=62;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.551
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1691,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.0398468203544617,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=62;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.0398468203544617
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5196,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=62;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Command R",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 701,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.738,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=63;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.738
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1702,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.5923334171772003,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=63;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.5923334171772003
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5247,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=63;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Command R Plus",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 30,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.671,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=2;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.671
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1031,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.3839432048797606,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=2;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.3839432048797606
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2136,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=2;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "DBRX Instruct",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 41,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.795,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=3;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.795
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1042,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.876643376111984,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=3;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.876643376111984
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2187,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=3;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "DeepSeek LLM Chat (67B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 52,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.94,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=4;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.94
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1053,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 9.76988450360298,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=4;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 9.76988450360298
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2238,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=4;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "DeepSeek v3",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 63,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.267,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=5;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.267
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1064,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 12.967224577903748,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=5;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 12.967224577903748
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2289,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=5;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Falcon (40B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 74,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.055,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=6;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.055
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1075,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.940216990470886,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=6;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.940216990470886
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2340,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=6;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Falcon (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 877,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.479,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=79;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.479
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1878,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.762208682537079,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=79;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.762208682537079
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6063,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=79;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-3.5 (text-davinci-002)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 866,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.615,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=78;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.615
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1867,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.199419307470322,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=78;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.199419307470322
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6012,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=78;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-3.5 (text-davinci-003)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 888,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.501,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=80;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.501
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1889,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 0.8983073465824127,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=80;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 0.8983073465824127
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6114,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=80;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-3.5 Turbo (0613)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 910,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.932,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=82;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.932
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1911,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.947624314308166,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=82;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.947624314308166
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6216,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=82;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-4 (0613)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 899,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.668,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=81;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.668
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1900,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.738402992963791,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=81;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.738402992963791
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6165,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=81;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-4 Turbo (1106 preview)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 921,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.824,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=83;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.824
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1922,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.91472976398468,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=83;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.91472976398468
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6267,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=83;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-4 Turbo (2024-04-09)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 932,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.905,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=84;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.905
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1933,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.227096201658249,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=84;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.227096201658249
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6318,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=84;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-4o (2024-05-13)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 943,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.909,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=85;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.909
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1944,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.9373713800907133,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=85;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.9373713800907133
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6369,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=85;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-4o (2024-08-06)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 954,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.843,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=86;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.843
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1955,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.5191967821121217,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=86;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.5191967821121217
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6420,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=86;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "GPT-4o mini (2024-07-18)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 712,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.816,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=64;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.816
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1713,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.513066102743149,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=64;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.513066102743149
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5298,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=64;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemini 1.0 Pro (002)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 734,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.785,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=66;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.785
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1735,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.7575640678405762,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=66;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.7575640678405762
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5400,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=66;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemini 1.5 Flash (001)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 756,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.328,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=68;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.328
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1757,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 0.8591284859287847,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=68;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 0.8591284859287847
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5502,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=68;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemini 1.5 Flash (002)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 723,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.836,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=65;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.836
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1724,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.205789808034897,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=65;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.205789808034897
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5349,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=65;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemini 1.5 Pro (001)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 745,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.817,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=67;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.817
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1746,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.1614130451679228,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=67;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.1614130451679228
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5451,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=67;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemini 1.5 Pro (002)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 767,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.946,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=69;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.946
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1768,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.4374724824428557,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=69;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.4374724824428557
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5553,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=69;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemini 2.0 Flash (Experimental)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 107,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.559,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=9;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.559
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1108,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.024561887741089,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=9;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.024561887741089
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2493,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=9;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemma (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 85,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.812,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=7;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.812
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1086,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.3315503742694856,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=7;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.3315503742694856
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2391,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=7;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemma 2 Instruct (27B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 96,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.762,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=8;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.762
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1097,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.720498773097992,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=8;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.720498773097992
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2442,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=8;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Gemma 2 Instruct (9B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 481,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.846,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=43;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.846
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1482,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.942030364751816,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=43;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.942030364751816
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4227,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=43;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Jamba 1.5 Large",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 470,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.691,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=42;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.691
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1471,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.8916997435092926,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=42;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.8916997435092926
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4176,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=42;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Jamba 1.5 Mini",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 459,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.67,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=41;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.67
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1460,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.8455032846927644,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=41;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.8455032846927644
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4125,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=41;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Jamba Instruct",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 437,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.159,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=39;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.159
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1438,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.417125414848328,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=39;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.417125414848328
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4023,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=39;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Jurassic-2 Grande (17B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 448,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.239,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=40;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.239
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1449,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.176425676584244,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=40;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.176425676584244
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4074,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=40;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Jurassic-2 Jumbo (178B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 239,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.489,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=21;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.489
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1240,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 12.338884568691254,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=21;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 12.338884568691254
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3105,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=21;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "LLaMA (65B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 118,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.266,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=10;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.266
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1119,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.7367573575973512,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=10;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.7367573575973512
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2544,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=10;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 2 (13B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 129,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.567,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=11;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.567
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1130,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.737159442663193,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=11;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.737159442663193
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2595,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=11;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 2 (70B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 140,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.154,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=12;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.154
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1141,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.95984334897995,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=12;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.95984334897995
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2646,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=12;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 2 (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 151,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.805,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=13;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.805
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1152,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.199564570903778,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=13;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.199564570903778
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2697,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=13;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3 (70B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 162,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.499,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=14;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.499
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1163,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.770608879327774,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=14;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.770608879327774
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2748,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=14;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3 (8B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 173,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.949,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=15;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.949
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1174,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.737115991592407,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=15;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.737115991592407
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2799,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=15;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3.1 Instruct Turbo (405B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 184,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.938,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=16;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.938
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1185,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.9902911036014554,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=16;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.9902911036014554
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2850,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=16;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3.1 Instruct Turbo (70B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 195,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.798,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=17;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.798
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1196,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.108796592712402,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=17;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.108796592712402
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2901,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=17;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3.1 Instruct Turbo (8B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 206,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.823,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=18;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.823
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1207,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.2738200931549073,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=18;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.2738200931549073
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2952,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=18;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3.2 Vision Instruct Turbo (11B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 217,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.936,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=19;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.936
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1218,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.8894128675460817,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=19;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.8894128675460817
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3003,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=19;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3.2 Vision Instruct Turbo (90B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 228,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.942,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=20;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.942
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1229,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.3539768285751344,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=20;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.3539768285751344
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3054,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=20;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Llama 3.3 Instruct Turbo (70B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 492,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.028,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=44;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.028
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1493,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 16.42652773284912,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=44;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 16.42652773284912
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4278,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=44;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Luminous Base (13B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 503,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.075,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=45;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.075
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1504,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 22.685439155817033,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=45;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 22.685439155817033
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4329,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=45;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Luminous Extended (30B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 514,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.137,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=46;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.137
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1515,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 48.241569149971006,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=46;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 48.241569149971006
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 4380,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=46;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Luminous Supreme (70B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 250,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.538,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=22;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.538
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1251,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.949965229511261,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=22;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.949965229511261
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3156,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=22;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral Instruct v0.3 (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 833,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.694,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=75;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.694
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1834,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 7.095049407720566,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=75;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 7.095049407720566
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5859,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=75;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral Large (2402)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 844,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.912,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=76;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.912
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1845,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.431343378543854,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=76;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.431343378543854
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5910,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=76;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral Large 2 (2407)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 822,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.706,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=74;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.706
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1823,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 9.718977437496186,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=74;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 9.718977437496186
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5808,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=74;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral Medium (2312)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 855,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.782,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=77;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.782
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1856,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.4254731934070588,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=77;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.4254731934070588
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5961,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=77;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral NeMo (2402)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 811,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.734,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=73;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.734
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1812,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.9720949590206147,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=73;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.9720949590206147
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5757,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=73;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral Small (2402)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 261,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.377,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=23;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.377
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1262,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.6323128745555877,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=23;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.6323128745555877
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3207,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=23;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mistral v0.1 (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 272,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.8,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=24;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.8
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1273,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.5390553929805755,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=24;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.5390553929805755
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3258,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=24;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mixtral (8x22B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 283,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.622,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=25;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.622
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1284,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.2728567245006563,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=25;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.2728567245006563
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3309,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=25;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Mixtral (8x7B 32K seqlen)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 294,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.044,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=26;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.044
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1295,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.4104921889305113,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=26;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.4104921889305113
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3360,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=26;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "OLMo (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 778,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.61,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=70;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.61
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1779,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.4403084371089936,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=70;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.4403084371089936
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5604,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=70;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "PaLM-2 (Bison)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 789,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.831,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=71;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.831
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1790,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.4373185629844665,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=71;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.4373185629844665
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5655,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=71;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "PaLM-2 (Unicorn)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 976,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.735,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=88;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.735
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1977,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.543274956703186,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=88;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.543274956703186
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6522,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=88;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Palmyra X V2 (33B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 987,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.831,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=89;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.831
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1988,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.069576686620712,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=89;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.069576686620712
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6573,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=89;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Palmyra X V3 (72B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 998,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.905,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=90;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.905
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1999,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 11.449529441833496,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=90;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 11.449529441833496
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6624,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=90;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Palmyra-X-004",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 305,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.581,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=27;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.581
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1306,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.1468114259243012,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=27;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.1468114259243012
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3411,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=27;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Phi-2",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 8,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.878,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=0;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.878
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1009,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 74.93269198083877,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=0;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 74.93269198083877
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2034,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=0;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Phi-3 (14B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 19,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": null,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=1;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": null
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1020,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": null,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=1;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": null
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 2085,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": null,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=1;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": null
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Phi-3 (7B)",
      "numericRowCount": 65,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 8,
        "reported": 65
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 327,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.693,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=29;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.693
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1328,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.965628466129303,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=29;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.965628466129303
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3513,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=29;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen1.5 (14B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 338,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.773,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=30;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.773
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1339,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.405816124200821,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=30;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.405816124200821
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3564,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=30;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen1.5 (32B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 349,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.799,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=31;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.799
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1350,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.5866835827827455,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=31;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.5866835827827455
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3615,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=31;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen1.5 (72B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 360,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=32;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1361,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.380831289768219,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=32;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.380831289768219
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3666,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=32;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen1.5 (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 316,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.815,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=28;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.815
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1317,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.537143226146698,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=28;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.537143226146698
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3462,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=28;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen1.5 Chat (110B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 371,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.92,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=33;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.92
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1372,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.592170278310776,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=33;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.592170278310776
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3717,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=33;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen2 Instruct (72B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 382,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.9,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=34;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.9
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1383,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.5583292784690856,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=34;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.5583292784690856
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3768,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=34;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen2.5 Instruct Turbo (72B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 393,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.83,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=35;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.83
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1394,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.7000067098140716,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=35;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.7000067098140716
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3819,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=35;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Qwen2.5 Instruct Turbo (7B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 965,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.871,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=87;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.871
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1966,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.6663423478603363,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=87;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.6663423478603363
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 6471,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=87;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Solar Pro",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 415,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.648,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=37;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.648
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1416,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.886563032150269,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=37;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.886563032150269
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3921,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=37;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Yi (34B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 426,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.375,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=38;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.375
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1427,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 1.8781680135726928,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=38;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 1.8781680135726928
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 3972,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=38;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Yi (6B)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "helm-gsm8k",
        "helm-legalbench",
        "helm-math",
        "helm-mean-win-rate",
        "helm-medqa",
        "helm-mmlu",
        "helm-narrativeqa",
        "helm-naturalquestions-closed-book",
        "helm-naturalquestions-open-book",
        "helm-openbookqa",
        "helm-wmt-2014"
      ],
      "examples": [
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 800,
          "metricId": "em",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.69,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Accuracy;row=72;column=GSM8K - EM",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.69
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 1801,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 13.45040065407753,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=Efficiency;row=72;column=GSM8K - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 13.45040065407753
        },
        {
          "artifact": "helm-lite/candidates.jsonl",
          "benchmarkId": "helm-gsm8k",
          "benchmarkName": "GSM8K · HELM Lite",
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "line": 5706,
          "metricId": "eval",
          "observedAt": "2025-01-10",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "lite",
            "release": "v1.13.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 1000.0,
          "sourceId": "helm-lite",
          "sourceLabel": "Stanford HELM · lite release artifact",
          "sourceLocator": "release=v1.13.0;table=General information;row=72;column=GSM8K - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 1000.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 73
      },
      "metricIds": [
        "bleu-4",
        "efficiency-score",
        "em",
        "equivalent-cot",
        "eval",
        "f1",
        "general-information-score",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated"
      ],
      "modelRef": "Yi Large (Preview)",
      "numericRowCount": 72,
      "observedAtMax": "2025-01-10",
      "observedAtMin": "2025-01-10",
      "rowCount": 73,
      "sourceId": "helm-lite",
      "sourceIds": [
        "helm-lite"
      ],
      "sourceLabels": [
        "Stanford HELM · lite release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "candidate": 1,
        "reported": 72
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-science_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 835,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.385",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=110;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 38.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 849,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.448",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=135;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.800000000000004
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 750,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.476",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=25;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 47.599999999999994
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 56
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "LLaMA-7B",
      "numericRowCount": 56,
      "observedAtMax": "2023-02-24",
      "observedAtMin": "2023-02-24",
      "rowCount": 56,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 56
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 837,
          "metricId": "Challenge score",
          "observedAt": "2023-04-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.363",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=112;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 36.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 850,
          "metricId": "Challenge score",
          "observedAt": "2023-04-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.434",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=136;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 740,
          "metricId": "Challenge score",
          "observedAt": "2023-04-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.445",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=15;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 54
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "falcon-7b",
      "numericRowCount": 54,
      "observedAtMax": "2023-04-24",
      "observedAtMin": "2023-04-24",
      "rowCount": 54,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 54
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-science_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 834,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.434",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=109;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 736,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.459",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=11;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 754,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.459",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=29;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 50
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Llama-2-7b",
      "numericRowCount": 50,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 50,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 50
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-multimodal",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 275,
          "metricId": "resolved",
          "observedAt": "2024-12-08",
          "protocol": {
            "checked": false,
            "harness": "Gru",
            "leaderboard_variant": "Lite",
            "scaffold": "Gru",
            "subject_type": "system"
          },
          "rawValue": 48.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[10]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 48.67
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 277,
          "metricId": "resolved",
          "observedAt": "2024-11-27",
          "protocol": {
            "checked": false,
            "harness": "Globant Code Fixer Agent",
            "leaderboard_variant": "Lite",
            "scaffold": "Globant Code Fixer Agent",
            "subject_type": "system"
          },
          "rawValue": 48.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[12]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 48.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 284,
          "metricId": "resolved",
          "observedAt": "2025-03-10",
          "protocol": {
            "checked": true,
            "harness": "CodeFuse-CGM",
            "leaderboard_variant": "Lite",
            "scaffold": "CodeFuse-CGM",
            "subject_type": "model"
          },
          "rawValue": 44.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[19]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 44.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 50
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Undisclosed",
      "numericRowCount": 50,
      "observedAtMax": "2025-06-10",
      "observedAtMin": "2024-05-09",
      "rowCount": 50,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 50
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "agents-last-exam"
      ],
      "examples": [
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 210,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "openclaw_cli",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.24093001414387707,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[104];split=full/full-spectrum;harness=openclaw_cli;model=ark-0614c;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 24.093001414387707
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 209,
          "metricId": "pass_rate",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "openclaw_cli",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.12727272727272726,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[104];split=full/full-spectrum;harness=openclaw_cli;model=ark-0614c;variant=unknown;metric=pass_rate",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 12.727272727272727
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 304,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db",
            "evaluator": "agents-last-exam",
            "harness": "claude-code",
            "harness_variant": null,
            "source_harness": "claude_code",
            "split": "full/last-exam",
            "split_benchmark_ref": "agents-last-exam-v1-full-last-exam",
            "split_track": "full/last-exam",
            "subject_type": "system"
          },
          "rawValue": 0.0993785955488623,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[151];split=full/last-exam;harness=claude_code;model=ark-0614c;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 9.93785955488623
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 48
      },
      "metricIds": [
        "avg_score",
        "pass_rate"
      ],
      "modelRef": "ark-0614c",
      "numericRowCount": 48,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 48,
      "sourceId": "agents-last-exam",
      "sourceIds": [
        "agents-last-exam"
      ],
      "sourceLabels": [
        "Agents' Last Exam · ALE-V1 leaderboard"
      ],
      "sourceUrls": [
        "https://agents-last-exam.org/api/demo/leaderboard"
      ],
      "statusCounts": {
        "reported": 48
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "agents-last-exam"
      ],
      "examples": [
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 224,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "list",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "grok_cli",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.1831974303030303,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[111];split=full/full-spectrum;harness=grok_cli;model=grok-4-3;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 18.31974303030303
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 223,
          "metricId": "pass_rate",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "list",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "grok_cli",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.07272727272727272,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[111];split=full/full-spectrum;harness=grok_cli;model=grok-4-3;variant=unknown;metric=pass_rate",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 7.2727272727272725
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 228,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db+list",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "openclaw",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.1354189575757576,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[113];split=full/full-spectrum;harness=openclaw;model=grok-4-3;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 13.541895757575759
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 48
      },
      "metricIds": [
        "avg_score",
        "pass_rate"
      ],
      "modelRef": "grok-4-3",
      "numericRowCount": 48,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 48,
      "sourceId": "agents-last-exam",
      "sourceIds": [
        "agents-last-exam"
      ],
      "sourceLabels": [
        "Agents' Last Exam · ALE-V1 leaderboard"
      ],
      "sourceUrls": [
        "https://agents-last-exam.org/api/demo/leaderboard"
      ],
      "statusCounts": {
        "reported": 48
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-science_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 737,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.494",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=12;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 49.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 755,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.603",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=30;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 765,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.494",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=40;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 49.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 48
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Llama-2-13b",
      "numericRowCount": 48,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 48,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 48
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 836,
          "metricId": "Challenge score",
          "observedAt": "2023-05-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.405",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=111;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 851,
          "metricId": "Challenge score",
          "observedAt": "2023-05-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.417",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=137;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.699999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 743,
          "metricId": "Challenge score",
          "observedAt": "2023-05-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.426",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=18;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 41
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "mpt-7b",
      "numericRowCount": 41,
      "observedAtMax": "2023-05-05",
      "observedAtMin": "2023-05-05",
      "rowCount": 41,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 41
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-science_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 751,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.527",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=26;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 780,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.527",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=55;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 791,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.527",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=66;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 40
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "LLaMA-13B",
      "numericRowCount": 40,
      "observedAtMax": "2023-02-24",
      "observedAtMin": "2023-02-24",
      "rowCount": 40,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 40
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 752,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.675",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=27;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 67.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 781,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.578",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=56;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 792,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.578",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=67;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 39
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "LLaMA-33B",
      "numericRowCount": 39,
      "observedAtMax": "2023-02-24",
      "observedAtMin": "2023-02-24",
      "rowCount": 39,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 39
      }
    },
    {
      "benchmarkCount": 31,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-algotune_external",
        "epoch-apex_agents_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-balrog_external",
        "epoch-chess_puzzles",
        "epoch-cl_bench_external",
        "epoch-critpt_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-frontiermath_tier_4",
        "epoch-gdpval_external",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-proofbench_external",
        "epoch-rli_external",
        "epoch-simplebench_external",
        "epoch-terminalbench_external",
        "epoch-vending_bench_2_external",
        "epoch-vpct_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "simpleqa",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 118,
          "metricId": "Performance",
          "observedAt": "2025-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1176.75",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=32;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1176.75
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 201,
          "metricId": "Score",
          "observedAt": "2025-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.83",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=5;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.83
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 239,
          "metricId": "Pass@1 score",
          "observedAt": "2025-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.315",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=25;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 31.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 38
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Accuracy mean",
        "Arena Score",
        "Average progress",
        "Correct",
        "ECI Score",
        "Overall",
        "Overall score",
        "Pass@1 score",
        "Performance",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "Win Rate (%)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gemini-3-pro-preview",
      "numericRowCount": 38,
      "observedAtMax": "2026-02-23",
      "observedAtMin": "2025-11-18",
      "rowCount": 38,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 38
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 177,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3968609865470852,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=29;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3968609865470852
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 585,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.0404999999999998,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=29;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.0404999999999998
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1547,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=29;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Amazon Nova Lite",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 183,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3834080717488789,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=30;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3834080717488789
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 591,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.1342376681614366,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=30;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.1342376681614366
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1572,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=30;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Amazon Nova Micro",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 165,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5179372197309418,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=27;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5179372197309418
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 573,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.7455403587443925,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=27;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.7455403587443925
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1497,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=27;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Amazon Nova Premier",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 171,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.4461883408071749,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=28;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.4461883408071749
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 579,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.947926008968607,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=28;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.947926008968607
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1522,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=28;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Amazon Nova Pro",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 189,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3632286995515695,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=31;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3632286995515695
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 597,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.329682314877018,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=31;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.329682314877018
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1597,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=31;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Claude 3.5 Haiku (20241022)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 195,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5650224215246636,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=32;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5650224215246636
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 603,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.261580738251519,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=32;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.261580738251519
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1622,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=32;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Claude 3.5 Sonnet (20241022)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 201,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6076233183856502,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=33;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6076233183856502
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 609,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.4586481999923295,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=33;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.4586481999923295
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1647,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=33;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Claude 3.7 Sonnet (20250219)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 207,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6434977578475336,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=34;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6434977578475336
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 615,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 13.452103998094396,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=34;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 13.452103998094396
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1672,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=34;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Claude 4 Sonnet (20250514)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 213,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.7062780269058296,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=35;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.7062780269058296
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 621,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 38.15993662211927,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=35;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 38.15993662211927
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1697,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=35;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Claude 4 Sonnet (20250514, extended thinking)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 45,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5381165919282511,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=7;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5381165919282511
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 453,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 74.37158904909553,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=7;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 74.37158904909553
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 997,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=7;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "DeepSeek v3",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 51,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.594170403587444,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=8;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.594170403587444
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 459,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 54.96293809649121,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=8;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 54.96293809649121
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1022,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=8;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "GLM-4.5-Air-FP8",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 315,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6591928251121076,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=52;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6591928251121076
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 723,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 9.906458986714282,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=52;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 9.906458986714282
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2122,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=52;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "GPT-4.1 (2025-04-14)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 321,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6143497757847534,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=53;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6143497757847534
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 729,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 8.216832675206822,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=53;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 8.216832675206822
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2147,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=53;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "GPT-4.1 mini (2025-04-14)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 327,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5067264573991032,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=54;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5067264573991032
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 735,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.816804544808084,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=54;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.816804544808084
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2172,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=54;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "GPT-4.1 nano (2025-04-14)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 303,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5201793721973094,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=50;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5201793721973094
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 711,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 13.64998589877056,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=50;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 13.64998589877056
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2072,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=50;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "GPT-4o (2024-11-20)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 309,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.36771300448430494,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=51;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.36771300448430494
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 717,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 8.813848996910814,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=51;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 8.813848996910814
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2097,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=51;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "GPT-4o mini (2024-07-18)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 249,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.437219730941704,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=41;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.437219730941704
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 657,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.7900896457278677,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=41;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.7900896457278677
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1847,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=41;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 1.5 Flash (002)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 243,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5336322869955157,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=40;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5336322869955157
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 651,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 7.392140488988081,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=40;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 7.392140488988081
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1822,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=40;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 1.5 Pro (002)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 255,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5560538116591929,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=42;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5560538116591929
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 663,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.919003446005919,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=42;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.919003446005919
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1872,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=42;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 2.0 Flash",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 261,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=43;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 669,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.372664878186623,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=43;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.372664878186623
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1897,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=43;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 2.0 Flash Lite (02-05 preview)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 273,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3901345291479821,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=45;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3901345291479821
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 681,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 38.125050564562336,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=45;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 38.125050564562336
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1947,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=45;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 2.5 Flash (04-17 preview)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 267,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3094170403587444,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=44;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3094170403587444
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 675,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 11.880136902022254,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=44;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 11.880136902022254
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1922,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=44;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 2.5 Flash-Lite",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 279,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.7488789237668162,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=46;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.7488789237668162
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 687,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 43.19425330858552,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=46;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 43.19425330858552
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1972,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=46;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 2.5 Pro (03-25 preview)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 285,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.8026905829596412,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=47;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.8026905829596412
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 693,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 69.16407415364355,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=47;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 69.16407415364355
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1997,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=47;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Gemini 3 Pro (Preview)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 393,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6502242152466368,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=65;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6502242152466368
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 801,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 30.88756059317311,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=65;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 30.88756059317311
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2447,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=65;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Grok 3 Beta",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 399,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6748878923766816,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=66;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6748878923766816
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 807,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 14.215015458419185,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=66;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 14.215015458419185
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2472,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=66;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Grok 3 mini Beta",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 405,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.726457399103139,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=67;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.726457399103139
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 813,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 223.96746500778625,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=67;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 223.96746500778625
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2497,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=67;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Grok 4 (0709)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 159,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3251121076233184,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=26;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3251121076233184
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 567,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.421983559569971,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=26;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.421983559569971
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1472,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=26;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "IBM Granite 3.3 8B Instruct",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 33,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3071748878923767,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=5;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3071748878923767
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 441,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.075281912970436,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=5;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.075281912970436
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 947,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=5;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "IBM Granite 4.0 Micro",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 39,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3834080717488789,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=6;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3834080717488789
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 447,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 17.606201725690354,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=6;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 17.606201725690354
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 972,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=6;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "IBM Granite 4.0 Small",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 69,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6524663677130045,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=11;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6524663677130045
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 477,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 50.10382581986654,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=11;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 50.10382581986654
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1097,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=11;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Kimi K2 Instruct",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 75,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5224215246636771,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=12;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5224215246636771
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 483,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 9.197324877362615,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=12;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 9.197324877362615
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1122,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=12;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Llama 3.1 Instruct Turbo (405B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 81,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.4260089686098655,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=13;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.4260089686098655
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 489,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 6.0952357684550265,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=13;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 6.0952357684550265
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1147,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=13;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Llama 3.1 Instruct Turbo (70B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 87,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.24663677130044842,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=14;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.24663677130044842
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 495,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.2803654104070277,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=14;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.2803654104070277
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1172,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=14;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Llama 3.1 Instruct Turbo (8B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 93,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6502242152466368,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=15;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6502242152466368
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 501,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 9.838454476921013,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=15;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 9.838454476921013
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1197,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=15;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Llama 4 Maverick (17Bx128E) Instruct FP8",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 99,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.5067264573991032,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=16;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.5067264573991032
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 507,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 11.026973943004693,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=16;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 11.026973943004693
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1222,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=16;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Llama 4 Scout (17Bx16E) Instruct",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 3,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.1681614349775785,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=0;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.1681614349775785
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 411,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 123.0189983149815,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=0;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 123.0189983149815
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 822,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=0;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Marin 8B Instruct",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 105,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.30269058295964124,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=17;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.30269058295964124
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 513,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 2.284658104849503,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=17;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 2.284658104849503
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1247,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=17;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Mistral Instruct v0.3 (7B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 297,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.4349775784753363,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=49;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.4349775784753363
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 705,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 12.217145950270341,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=49;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 12.217145950270341
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2047,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=49;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Mistral Large (2411)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 291,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.3923766816143498,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=48;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.3923766816143498
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 699,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 5.049520614435854,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=48;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 5.049520614435854
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2022,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=48;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Mistral Small 3.1 (2503)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 111,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.33408071748878926,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=18;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.33408071748878926
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 519,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 4.760301354220095,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=18;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 4.760301354220095
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1272,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=18;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Mixtral Instruct (8x22B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 117,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.29596412556053814,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=19;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.29596412556053814
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 525,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.1633052681593616,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=19;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.1633052681593616
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1297,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=19;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Mixtral Instruct (8x7B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 15,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.31614349775784756,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=2;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.31614349775784756
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 423,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 44.36780591235567,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=2;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 44.36780591235567
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 872,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=2;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "OLMo 2 13B Instruct November 2024",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 9,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.28699551569506726,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=1;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.28699551569506726
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 417,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 161.24673478646127,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=1;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 161.24673478646127
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 847,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=1;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "OLMo 2 32B Instruct March 2025",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 21,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.29596412556053814,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=3;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.29596412556053814
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 429,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 184.73346061877606,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=3;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 184.73346061877606
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 897,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=3;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "OLMo 2 7B Instruct November 2024",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 27,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.21973094170403587,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=4;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.21973094170403587
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 435,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 263.9177615305768,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=4;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 263.9177615305768
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 922,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=4;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "OLMoE 1B-7B Instruct January 2025",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 387,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.42152466367713004,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=64;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.42152466367713004
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 795,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 14.42766729758994,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=64;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 14.42766729758994
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2422,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=64;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Palmyra Fin",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 381,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.36771300448430494,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=63;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.36771300448430494
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 789,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 0.3557077256018805,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=63;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 0.3557077256018805
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2397,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=63;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Palmyra Med",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 375,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6614349775784754,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=62;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6614349775784754
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 783,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 9.251234515365464,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=62;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 9.251234515365464
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2372,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=62;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Palmyra X5",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 369,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.39461883408071746,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=61;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.39461883408071746
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 777,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 20.444375363700594,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=61;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 20.444375363700594
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2347,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=61;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Palmyra-X-004",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 123,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.4260089686098655,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=20;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.4260089686098655
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 531,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 28.71905704036422,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=20;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 28.71905704036422
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1322,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=20;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Qwen2.5 Instruct Turbo (72B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 129,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.34080717488789236,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=21;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.34080717488789236
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 537,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 3.4745728910771185,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=21;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 3.4745728910771185
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1347,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=21;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Qwen2.5 Instruct Turbo (7B)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 135,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6233183856502242,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=22;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6233183856502242
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 543,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 237.41318658488748,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=22;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 237.41318658488748
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1372,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=22;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Qwen3 235B A22B FP8 Throughput",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 141,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.726457399103139,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=23;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.726457399103139
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 549,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 103.30346254970995,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=23;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 103.30346254970995
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1397,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=23;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Qwen3 235B A22B Instruct 2507 FP8",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 147,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6300448430493274,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=24;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6300448430493274
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 555,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 40.06039341950096,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=24;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 40.06039341950096
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1422,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=24;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "Qwen3-Next 80B A3B Thinking",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 57,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.6838565022421524,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=9;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.6838565022421524
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 465,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 18.8192116278704,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=9;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 18.8192116278704
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1047,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=9;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "gpt-oss-120b",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 63,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.594170403587444,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=10;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.594170403587444
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 471,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 27.56541810923093,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=10;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 27.56541810923093
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 1072,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=10;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "gpt-oss-20b",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 357,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.7533632286995515,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=59;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.7533632286995515
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 765,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 48.0242628821343,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=59;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 48.0242628821343
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2297,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=59;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "o3 (2025-04-16)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "helm-gpqa",
        "helm-ifeval",
        "helm-mean-score",
        "helm-mmlu-pro",
        "helm-omni-math",
        "helm-wildbench"
      ],
      "examples": [
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 363,
          "metricId": "cot-correct",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "higher",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Accuracy"
          },
          "rawValue": 0.7354260089686099,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Accuracy;row=60;column=GPQA - COT correct",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "fraction",
          "value": 0.7354260089686099
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 771,
          "metricId": "observed-inference-time-s",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "Efficiency"
          },
          "rawValue": 22.412139415206397,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=Efficiency;row=60;column=GPQA - Observed inference time (s)",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "seconds",
          "value": 22.412139415206397
        },
        {
          "artifact": "helm-capabilities/candidates.jsonl",
          "benchmarkId": "helm-gpqa",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "line": 2322,
          "metricId": "eval",
          "observedAt": "2025-11-24",
          "protocol": {
            "direction": "lower",
            "harness": "helm",
            "project": "capabilities",
            "release": "v1.15.0",
            "source_status": "published_release",
            "table": "General information"
          },
          "rawValue": 446.0,
          "sourceId": "helm-capabilities",
          "sourceLabel": "Stanford HELM · capabilities release artifact",
          "sourceLocator": "release=v1.15.0;table=General information;row=60;column=GPQA - # eval",
          "sourceUrl": "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json",
          "unit": "count",
          "value": 446.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 37
      },
      "metricIds": [
        "acc",
        "cot-correct",
        "efficiency-score",
        "eval",
        "ifeval-strict-acc",
        "observed-inference-time-s",
        "output-tokens",
        "prompt-tokens",
        "score",
        "train",
        "truncated",
        "wb-score"
      ],
      "modelRef": "o4-mini (2025-04-16)",
      "numericRowCount": 37,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 37,
      "sourceId": "helm-capabilities",
      "sourceIds": [
        "helm-capabilities"
      ],
      "sourceLabels": [
        "Stanford HELM · capabilities release artifact"
      ],
      "sourceUrls": [
        "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
      ],
      "statusCounts": {
        "reported": 37
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-common_sense_qa_2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 739,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.574",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=14;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 756,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.783",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=31;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 78.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 786,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.574",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=61;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 36
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Llama-2-70b-hf",
      "numericRowCount": 36,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 36,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 36
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 753,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.695",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=28;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 69.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 782,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.56",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=57;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.00000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 793,
          "metricId": "Challenge score",
          "observedAt": "2023-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.56",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=68;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.00000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 35
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "LLaMA-65B",
      "numericRowCount": 35,
      "observedAtMax": "2023-02-24",
      "observedAtMin": "2023-02-24",
      "rowCount": 35,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 35
      }
    },
    {
      "benchmarkCount": 25,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-apex_agents_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-balrog_external",
        "epoch-blueprint_bench_2_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-frontiermath_tier_4",
        "epoch-gdp_pdf_external",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-proofbench_external",
        "epoch-simplebench_external",
        "epoch-terminalbench_external",
        "epoch-vending_bench_2_external",
        "epoch-vpct_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 104,
          "metricId": "Performance",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1367.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=18;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1367.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 243,
          "metricId": "Pass@1 score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.24",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=29;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 24.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 383,
          "metricId": "Score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1278",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=107;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.1278
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 34
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Accuracy mean",
        "Arena Score",
        "Average progress",
        "Correct",
        "ECI Score",
        "GDP.pdf score",
        "Overall score",
        "Pass@1 score",
        "Performance",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "mean_score"
      ],
      "modelRef": "gemini-3-flash-preview",
      "numericRowCount": 34,
      "observedAtMax": "2026-03-06",
      "observedAtMin": "06/06/2026",
      "rowCount": 34,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 34
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 276,
          "metricId": "resolved",
          "observedAt": "2025-06-13",
          "protocol": {
            "checked": false,
            "harness": "ExpeRepair-v1.0",
            "leaderboard_variant": "Lite",
            "scaffold": "ExpeRepair-v1.0",
            "subject_type": "system"
          },
          "rawValue": 48.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[11]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 48.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 280,
          "metricId": "resolved",
          "observedAt": "2025-02-05",
          "protocol": {
            "checked": false,
            "harness": "DARS Agent",
            "leaderboard_variant": "Lite",
            "scaffold": "DARS Agent",
            "subject_type": "system"
          },
          "rawValue": 47.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[15]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 47.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 266,
          "metricId": "resolved",
          "observedAt": "2025-04-25",
          "protocol": {
            "checked": false,
            "harness": "Refact.ai Agent",
            "leaderboard_variant": "Lite",
            "scaffold": "Refact.ai Agent",
            "subject_type": "system"
          },
          "rawValue": 60.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[1]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 60.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 33
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Multiple",
      "numericRowCount": 33,
      "observedAtMax": "2025-11-03",
      "observedAtMin": "2024-08-28",
      "rowCount": 33,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 33
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 741,
          "metricId": "Challenge score",
          "observedAt": "2023-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.567",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=16;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.699999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 773,
          "metricId": "Challenge score",
          "observedAt": "2023-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5469",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=48;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.690000000000005
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 774,
          "metricId": "Challenge score",
          "observedAt": "2023-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6186",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=49;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.86000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 31
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "falcon-40b",
      "numericRowCount": 31,
      "observedAtMax": "2023-03-15",
      "observedAtMin": "2023-03-15",
      "rowCount": 31,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 31
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 5,
          "metricId": "Score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.471",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=9;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.471
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 768,
          "metricId": "Challenge score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.555",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=43;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.50000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2401.04088",
          "line": 802,
          "metricId": "Challenge score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.549",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=77;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.900000000000006
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 30
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Mistral-7B-v0.1",
      "numericRowCount": 30,
      "observedAtMax": "2023-09-27",
      "observedAtMin": "2023-09-27",
      "rowCount": 30,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 30
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-multimodal",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 298,
          "metricId": "resolved",
          "observedAt": "2024-06-27",
          "protocol": {
            "checked": false,
            "harness": "AbanteAI MentatBot",
            "leaderboard_variant": "Lite",
            "scaffold": "AbanteAI MentatBot",
            "subject_type": "system"
          },
          "rawValue": 38.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[33]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 38.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 306,
          "metricId": "resolved",
          "observedAt": "2024-07-23",
          "protocol": {
            "checked": false,
            "harness": "Bytedance MarsCode Agent",
            "leaderboard_variant": "Lite",
            "scaffold": "Bytedance MarsCode Agent",
            "subject_type": "system"
          },
          "rawValue": 34.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[41]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 34.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 309,
          "metricId": "resolved",
          "observedAt": "2024-10-28",
          "protocol": {
            "checked": false,
            "harness": "Agentless-1.5",
            "leaderboard_variant": "Lite",
            "scaffold": "Agentless-1.5",
            "subject_type": "system"
          },
          "rawValue": 32.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[44]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 32.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 30
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4o",
      "numericRowCount": 30,
      "observedAtMax": "2025-05-31",
      "observedAtMin": "2024-06-12",
      "rowCount": 30,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 30
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-multimodal",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 279,
          "metricId": "resolved",
          "observedAt": "2024-11-22",
          "protocol": {
            "checked": false,
            "harness": "devlo",
            "leaderboard_variant": "Lite",
            "scaffold": "devlo",
            "subject_type": "system"
          },
          "rawValue": 47.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[14]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 47.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 281,
          "metricId": "resolved",
          "observedAt": "2025-06-19",
          "protocol": {
            "checked": false,
            "harness": "KGCompass",
            "leaderboard_variant": "Lite",
            "scaffold": "KGCompass",
            "subject_type": "system"
          },
          "rawValue": 46.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[16]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 46.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 283,
          "metricId": "resolved",
          "observedAt": "2024-12-07",
          "protocol": {
            "checked": false,
            "harness": "Kodu-v1",
            "leaderboard_variant": "Lite",
            "scaffold": "Kodu-v1",
            "subject_type": "system"
          },
          "rawValue": 44.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[18]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 44.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 29
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 3.5 Sonnet",
      "numericRowCount": 29,
      "observedAtMax": "2025-09-11",
      "observedAtMin": "2024-06-20",
      "rowCount": 29,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 29
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 735,
          "metricId": "Challenge score",
          "observedAt": "2022-04-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.53",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=10;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2303.08774",
          "line": 842,
          "metricId": "Challenge score",
          "observedAt": "2022-04-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.852",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=122;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 85.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2305.10403",
          "line": 761,
          "metricId": "Challenge score",
          "observedAt": "2022-04-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.601",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=36;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.099999999999994
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 28
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "PaLM 540B",
      "numericRowCount": 28,
      "observedAtMax": "2022-04-04",
      "observedAtMin": "2022-04-04",
      "rowCount": 28,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 28
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 738,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.545",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=13;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.50000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 766,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.545",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=41;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.50000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 785,
          "metricId": "Challenge score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.545",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=60;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.50000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 26
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "Llama-2-34b",
      "numericRowCount": 26,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 26,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 26
      }
    },
    {
      "benchmarkCount": 20,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-apex_agents_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-balrog_external",
        "epoch-chess_puzzles",
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-frontiermath_tier_4",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-terminalbench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 27,
          "metricId": "Percent correct",
          "observedAt": "2025-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "79.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=14;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 258,
          "metricId": "Pass@1 score",
          "observedAt": "2025-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.152",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=44;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 378,
          "metricId": "Score",
          "observedAt": "2025-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.15975",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=102;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.15975
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 25
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Accuracy mean",
        "Average progress",
        "Average score",
        "ECI Score",
        "Mean score",
        "Overall score",
        "Pass@1 score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "grok-4-0709",
      "numericRowCount": 25,
      "observedAtMax": "2025-11-03",
      "observedAtMin": "2025-07-09",
      "rowCount": 25,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 25
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "agents-last-exam"
      ],
      "examples": [
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 312,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db+list",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "cursor_cli",
            "split": "full/last-exam",
            "split_benchmark_ref": "agents-last-exam-v1-full-last-exam",
            "split_track": "full/last-exam",
            "subject_type": "system"
          },
          "rawValue": 0.08756310526315789,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[155];split=full/last-exam;harness=cursor_cli;model=composer-2-5;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 8.756310526315788
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 311,
          "metricId": "pass_rate",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db+list",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "cursor_cli",
            "split": "full/last-exam",
            "split_benchmark_ref": "agents-last-exam-v1-full-last-exam",
            "split_track": "full/last-exam",
            "subject_type": "system"
          },
          "rawValue": 0,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[155];split=full/last-exam;harness=cursor_cli;model=composer-2-5;variant=unknown;metric=pass_rate",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 408,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db+list",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "cursor_cli",
            "split": "full/near-term",
            "split_benchmark_ref": "agents-last-exam-v1-full-near-term",
            "split_track": "full/near-term",
            "subject_type": "system"
          },
          "rawValue": 0.6229800382758721,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[203];split=full/near-term;harness=cursor_cli;model=composer-2-5;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 62.29800382758721
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 24
      },
      "metricIds": [
        "avg_score",
        "pass_rate"
      ],
      "modelRef": "composer-2-5",
      "numericRowCount": 24,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 24,
      "sourceId": "agents-last-exam",
      "sourceIds": [
        "agents-last-exam"
      ],
      "sourceLabels": [
        "Agents' Last Exam · ALE-V1 leaderboard"
      ],
      "sourceUrls": [
        "https://agents-last-exam.org/api/demo/leaderboard"
      ],
      "statusCounts": {
        "reported": 24
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "agents-last-exam"
      ],
      "examples": [
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 230,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "grok_cli",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.12991377341047364,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[114];split=full/full-spectrum;harness=grok_cli;model=grok-3;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 12.991377341047365
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 229,
          "metricId": "pass_rate",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "grok_cli",
            "split": "full/full-spectrum",
            "split_benchmark_ref": "agents-last-exam-v1-full-full-spectrum",
            "split_track": "full/full-spectrum",
            "subject_type": "system"
          },
          "rawValue": 0.05454545454545454,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[114];split=full/full-spectrum;harness=grok_cli;model=grok-3;variant=unknown;metric=pass_rate",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 5.454545454545454
        },
        {
          "artifact": "agents-last-exam/candidates.jsonl",
          "benchmarkId": "agents-last-exam",
          "benchmarkName": "Agents' Last Exam",
          "evidenceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "line": 346,
          "metricId": "avg_score",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "ALE-v1",
            "benchmark_version_id": "agents-last-exam@v1",
            "cost_source": "db",
            "evaluator": "agents-last-exam",
            "harness": "agents-last-exam",
            "harness_variant": null,
            "source_harness": "grok_cli",
            "split": "full/last-exam",
            "split_benchmark_ref": "agents-last-exam-v1-full-last-exam",
            "split_track": "full/last-exam",
            "subject_type": "system"
          },
          "rawValue": 0.018223973684210526,
          "sourceId": "agents-last-exam",
          "sourceLabel": "Agents' Last Exam · ALE-V1 leaderboard",
          "sourceLocator": "rows[172];split=full/last-exam;harness=grok_cli;model=grok-3;variant=unknown;metric=avg_score",
          "sourceUrl": "https://agents-last-exam.org/api/demo/leaderboard",
          "unit": "percent",
          "value": 1.8223973684210526
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 24
      },
      "metricIds": [
        "avg_score",
        "pass_rate"
      ],
      "modelRef": "grok-3",
      "numericRowCount": 24,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 24,
      "sourceId": "agents-last-exam",
      "sourceIds": [
        "agents-last-exam"
      ],
      "sourceLabels": [
        "Agents' Last Exam · ALE-V1 leaderboard"
      ],
      "sourceUrls": [
        "https://agents-last-exam.org/api/demo/leaderboard"
      ],
      "statusCounts": {
        "reported": 24
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 1088,
          "metricId": "Score",
          "observedAt": "2021-12-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.793",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=148;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.793
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1111,
          "metricId": "Score",
          "observedAt": "2021-12-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.794",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=175;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.794
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2203.15556",
          "line": 1124,
          "metricId": "Score",
          "observedAt": "2021-12-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.793",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=188;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.793
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 24
      },
      "metricIds": [
        "Accuracy",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Gopher (280B)",
      "numericRowCount": 24,
      "observedAtMax": "2021-12-08",
      "observedAtMin": "2021-12-08",
      "rowCount": 24,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 24
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.11990",
          "line": 11,
          "metricId": "Score",
          "observedAt": "2022-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.366",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=15;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.366
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.11990",
          "line": 13,
          "metricId": "Score",
          "observedAt": "2022-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.397",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=17;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.397
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.11990",
          "line": 15,
          "metricId": "Score",
          "observedAt": "2022-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.396",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=19;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.396
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 24
      },
      "metricIds": [
        "Accuracy",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Megatron-Turing NLG 530B",
      "numericRowCount": 24,
      "observedAtMax": "2022-01-28",
      "observedAtMin": "2022-01-28",
      "rowCount": 24,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 24
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 6,
          "metricId": "Score",
          "observedAt": "2024-02-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.487",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=10;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.487
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 769,
          "metricId": "Challenge score",
          "observedAt": "2024-02-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.532",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=44;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 809,
          "metricId": "Challenge score",
          "observedAt": "2024-02-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.783",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=84;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 78.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 24
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "gemma-7b",
      "numericRowCount": 24,
      "observedAtMax": "2024-02-21",
      "observedAtMin": "2024-02-21",
      "rowCount": 24,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 24
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 369,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "97.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=16;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 97.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 370,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "65.217",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=16;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 65.217
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 371,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "74.648",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=16;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 74.648
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "grok-4.3",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 415,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "66.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=18;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 66.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 416,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "67.391",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=18;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 67.391
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 417,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "63.38",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=18;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 63.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "grok-build-0.1",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 714,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "99.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=31;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 99.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 715,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "67.391",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=31;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 67.391
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 716,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "74.648",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=31;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 74.648
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inkling-xhigh",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 668,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "82.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=29;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 82.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 669,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "78.261",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=29;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 78.261
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 670,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "76.056",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=29;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 76.056
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "muse-spark-1.1-xhigh",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 875,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "89.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=38;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 89.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 876,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "80.435",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=38;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 80.435
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 877,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "74.648",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=38;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 74.648
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "muse-spark-1.2-xhigh",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 1013,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "98.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=44;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 98.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 1014,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "78.261",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=44;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 78.261
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 1015,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "73.239",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=44;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 73.239
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ox-alpha-max",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "livebench-amps-hard",
        "livebench-code-completion",
        "livebench-code-generation",
        "livebench-connections",
        "livebench-consecutive-events",
        "livebench-integrals-with-game",
        "livebench-javascript",
        "livebench-logic-with-navigation",
        "livebench-math-comp",
        "livebench-olympiad",
        "livebench-paraphrase",
        "livebench-plot-unscrambling",
        "livebench-python",
        "livebench-simplify",
        "livebench-spatial",
        "livebench-story-generation",
        "livebench-summarize",
        "livebench-tablejoin",
        "livebench-tablereformat",
        "livebench-theory-of-mind",
        "livebench-typescript",
        "livebench-typos",
        "livebench-zebra-puzzle"
      ],
      "examples": [
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-amps-hard",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 898,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "AMPS_Hard"
          },
          "rawValue": "98.0",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=39;column=AMPS_Hard",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 98.0
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-completion",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 899,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_completion"
          },
          "rawValue": "80.435",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=39;column=code_completion",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 80.435
        },
        {
          "artifact": "livebench-official/candidates.jsonl",
          "benchmarkId": "livebench-code-generation",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "line": 900,
          "metricId": "score",
          "observedAt": "2026_06_25",
          "protocol": {
            "harness": "livebench-official-table",
            "release_date": "2026_06_25",
            "task": "code_generation"
          },
          "rawValue": "84.507",
          "sourceId": "livebench-official",
          "sourceLabel": "LiveBench · official dated release table",
          "sourceLocator": "path=public/table_2026_06_25.csv;row=39;column=code_generation",
          "sourceUrl": "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv",
          "unit": "percent",
          "value": 84.507
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "smaug-agentic",
      "numericRowCount": 23,
      "observedAtMax": "2026_06_25",
      "observedAtMin": "2026_06_25",
      "rowCount": 23,
      "sourceId": "livebench-official",
      "sourceIds": [
        "livebench-official"
      ],
      "sourceLabels": [
        "LiveBench · official dated release table"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 19,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-the_agent_company_external",
        "epoch-vpct_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench",
        "osworld",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 67,
          "metricId": "Percent correct",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "60.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=57;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 476,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=200;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 701,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.136",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=212;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.136
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "% Score",
        "ACW Avg Score",
        "Accuracy",
        "Correct",
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219",
      "numericRowCount": 23,
      "observedAtMax": "2025-06-14",
      "observedAtMin": "2025-02-24",
      "rowCount": 23,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 23,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-apex_agents_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-critpt_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-gdpval_external",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-the_agent_company_external",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "livebench",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 53,
          "metricId": "Percent correct",
          "observedAt": "2024-12-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "18.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=41;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 18.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 277,
          "metricId": "Pass@1 score",
          "observedAt": "2024-11-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.011000000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=63;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 490,
          "metricId": "Score",
          "observedAt": "2024-11-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=214;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 23
      },
      "metricIds": [
        "% Score",
        "ACW Avg Score",
        "Accuracy",
        "Correct",
        "ECI Score",
        "EM",
        "Global average",
        "Mean score",
        "Pass@1 score",
        "Percent correct",
        "Score",
        "Score OPT@1",
        "Win Rate (%)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4o-2024-11-20",
      "numericRowCount": 23,
      "observedAtMax": "2024-12-30",
      "observedAtMin": "2024-11-20",
      "rowCount": 23,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 23
      }
    },
    {
      "benchmarkCount": 21,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-balrog_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-frontiermath_tier_4",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-the_agent_company_external",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 54,
          "metricId": "Percent correct",
          "observedAt": "2025-01-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "51.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=42;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 51.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 873,
          "metricId": "Average progress",
          "observedAt": "2024-10-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.326",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=14;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1705,
          "metricId": "Accuracy",
          "observedAt": "2024-10-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0091",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=36;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.91
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 21
      },
      "metricIds": [
        "% Score",
        "ACW Avg Score",
        "Accuracy",
        "Average progress",
        "Correct",
        "ECI Score",
        "EM",
        "Global average",
        "Mean score",
        "Overall score",
        "Percent correct",
        "Score (AVG@5)",
        "Score OPT@1",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-3-5-sonnet-20241022",
      "numericRowCount": 21,
      "observedAtMax": "2025-01-17",
      "observedAtMin": "2024-10-22",
      "rowCount": 21,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 21
      }
    },
    {
      "benchmarkCount": 21,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-ale_bench_external",
        "epoch-apex_agents_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-cl_bench_external",
        "epoch-critpt_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 51,
          "metricId": "Percent correct",
          "observedAt": "2025-06-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "81.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=38;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 81.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 133,
          "metricId": "Performance",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "933.55",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=47;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 933.55
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 255,
          "metricId": "Pass@1 score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.172",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=41;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 17.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 21
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Overall",
        "Pass@1 score",
        "Percent correct",
        "Performance",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "mean_score"
      ],
      "modelRef": "o3-2025-04-16_high",
      "numericRowCount": 21,
      "observedAtMax": "2025-06-25",
      "observedAtMin": "2025-04-16",
      "rowCount": 21,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 21
      }
    },
    {
      "benchmarkCount": 21,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-ale_bench_external",
        "epoch-algotune_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-gdpval_external",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 50,
          "metricId": "Percent correct",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "72.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=37;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 72.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 141,
          "metricId": "Performance",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "826.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=55;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 826.17
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 203,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.72",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=7;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.72
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 21
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Performance",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "Win Rate (%)",
        "mean_score"
      ],
      "modelRef": "o4-mini-2025-04-16_high",
      "numericRowCount": 21,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 21,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 21
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-document",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-document",
          "benchmarkName": "Arena Document",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2554,
          "metricId": "arena_score_bt",
          "observedAt": "2026-07-30",
          "protocol": {
            "arena_config": "document",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1413.3172092037516,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=document;split=latest;row_idx=33",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1413.3172092037516
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 25,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1465.9720512404658,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=24",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1465.9720512404658
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 425,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1513.293525825947,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=424",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1513.293525825947
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 20
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3-flash",
      "numericRowCount": 20,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-01-09",
      "rowCount": 20,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 20
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-document",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-document",
          "benchmarkName": "Arena Document",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2547,
          "metricId": "arena_score_bt",
          "observedAt": "2026-07-30",
          "protocol": {
            "arena_config": "document",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1433.0110891814002,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=document;split=latest;row_idx=26",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1433.0110891814002
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 13,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1479.352753875612,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=12",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1479.352753875612
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 412,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1528.6570936664382,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=411",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1528.6570936664382
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 20
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3-pro",
      "numericRowCount": 20,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-01-09",
      "rowCount": 20,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 20
      }
    },
    {
      "benchmarkCount": 18,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-hella_swag_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 59,
          "metricId": "Percent correct",
          "observedAt": "2024-12-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "48.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=47;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 48.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 797,
          "metricId": "Challenge score",
          "observedAt": "2024-12-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.953",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=72;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 95.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2412.19437",
          "line": 908,
          "metricId": "Average",
          "observedAt": "2024-12-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.829",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=13;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.829
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 20
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Global average",
        "Overall accuracy",
        "Overall score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "DeepSeek-V3",
      "numericRowCount": 20,
      "observedAtMax": "2024-12-26",
      "observedAtMin": "2024-12-26",
      "rowCount": 20,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 20
      }
    },
    {
      "benchmarkCount": 18,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-deepresearchbench_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-gdpval_external",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vpct_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "osworld",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 23,
          "metricId": "Percent correct",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "76.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=10;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 76.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 424,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0298",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=148;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0298
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 625,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5383",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=135;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.5383
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 20
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Average score",
        "Correct",
        "ECI Score",
        "Mean score",
        "Percent correct",
        "Score",
        "Win Rate (%)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "o3-2025-04-16_medium",
      "numericRowCount": 20,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 20,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 20
      }
    },
    {
      "benchmarkCount": 18,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-balrog_external",
        "epoch-bool_q_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-gsm8k_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-piqa_external",
        "epoch-simplebench_external",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 64,
          "metricId": "Percent correct",
          "observedAt": "2024-12-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "3.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=53;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 477,
          "metricId": "Score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=201;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 886,
          "metricId": "Average progress",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.174",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=27;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 17.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 19
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Average progress",
        "Correct",
        "ECI Score",
        "EM",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gpt-4o-mini-2024-07-18",
      "numericRowCount": 19,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-07-18",
      "rowCount": 19,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 19
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 458,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1475.5691792231303,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=457",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1475.5691792231303
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 56,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1442.2864822938236,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=55",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1442.2864822938236
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 863,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1436.7998822699296,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=862",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1436.7998822699296
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 18
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3-flash (thinking-minimal)",
      "numericRowCount": 18,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-08-21",
      "rowCount": 18,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 18
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 116,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1414.7744952462408,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=115",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1414.7744952462408
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 486,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1462.833851584788,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=485",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1462.833851584788
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 914,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1399.0787896246925,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=913",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1399.0787896246925
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 18
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3.1-flash-lite-preview",
      "numericRowCount": 18,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-08-21",
      "rowCount": 18,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 18
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-document",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-document",
          "benchmarkName": "Arena Document",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2549,
          "metricId": "arena_score_bt",
          "observedAt": "2026-07-30",
          "protocol": {
            "arena_config": "document",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1423.6972038163226,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=document;split=latest;row_idx=28",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1423.6972038163226
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 468,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1470.8380090779501,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=467",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1470.8380090779501
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 58,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1441.697076311538,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=57",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1441.697076311538
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 18
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-4-31b",
      "numericRowCount": 18,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-07-30",
      "rowCount": 18,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 18
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-search",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2509,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1164.8004512187958,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=22",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1164.8004512187958
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 141,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1397.485203920536,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=140",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1397.485203920536
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 529,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1429.0364041323858,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=528",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1429.0364041323858
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 18
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4.3",
      "numericRowCount": 18,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-08-21",
      "rowCount": 18,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 18
      }
    },
    {
      "benchmarkCount": 16,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-geobench_external",
        "epoch-gso_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-the_agent_company_external",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "osworld"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 34,
          "metricId": "Percent correct",
          "observedAt": "2025-05-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "56.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=21;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 452,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.012700000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=176;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.012700000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 678,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2383",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=189;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.2383
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 18
      },
      "metricIds": [
        "% Score",
        "ACW Avg Score",
        "Accuracy",
        "Correct",
        "ECI Score",
        "Mean score",
        "Percent correct",
        "Score",
        "Score OPT@1",
        "mean_score"
      ],
      "modelRef": "claude-sonnet-4-20250514",
      "numericRowCount": 18,
      "observedAtMax": "2025-06-14",
      "observedAtMin": "2025-05-22",
      "rowCount": 18,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 18
      }
    },
    {
      "benchmarkCount": 18,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-ale_bench_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-frontiermath_tier_4",
        "epoch-geobench_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 45,
          "metricId": "Percent correct",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "52.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=32;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 174,
          "metricId": "Performance",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "558.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=88;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 558.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 472,
          "metricId": "Score",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0042",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=196;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0042
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 18
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Overall score",
        "Percent correct",
        "Performance",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gpt-4.1-2025-04-14",
      "numericRowCount": 18,
      "observedAtMax": "2025-04-14",
      "observedAtMin": "2025-04-14",
      "rowCount": 18,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 18
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 459,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1475.398565204057,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=458",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1475.398565204057
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 77,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1434.6212254308346,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=76",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1434.6212254308346
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 852,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.582600177772,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=851",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.582600177772
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-4-26b-a4b",
      "numericRowCount": 17,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 17,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-document",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-document",
          "benchmarkName": "Arena Document",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2553,
          "metricId": "arena_score_bt",
          "observedAt": "2026-07-30",
          "protocol": {
            "arena_config": "document",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1413.906463381873,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=document;split=latest;row_idx=32",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1413.906463381873
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 39,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1450.8823763174714,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=38",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1450.8823763174714
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 453,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1482.5091697595087,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=452",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1482.5091697595087
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4.20-beta-0309-reasoning",
      "numericRowCount": 17,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-07-30",
      "rowCount": 17,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-document",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-document",
          "benchmarkName": "Arena Document",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2533,
          "metricId": "arena_score_bt",
          "observedAt": "2026-07-30",
          "protocol": {
            "arena_config": "document",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1471.6610258406463,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=document;split=latest;row_idx=12",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1471.6610258406463
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 14,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1478.3224379541007,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=13",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1478.3224379541007
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 414,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1527.8111614483698,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=413",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1527.8111614483698
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "muse-spark-1.1",
      "numericRowCount": 17,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-07-30",
      "rowCount": 17,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 749,
          "metricId": "Challenge score",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.38",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=24;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 38.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 913,
          "metricId": "Average",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.488",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=18;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.488
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 931,
          "metricId": "Average",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4878",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=38;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4878
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Baichuan-2-13B-Base",
      "numericRowCount": 17,
      "observedAtMax": "2023-09-06",
      "observedAtMin": "2023-09-06",
      "rowCount": 17,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 17,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-algotune_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-balrog_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 58,
          "metricId": "Percent correct",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "56.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=46;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 204,
          "metricId": "Score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=8;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 451,
          "metricId": "Score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.013000000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=175;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.013000000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "Accuracy",
        "Average progress",
        "ECI Score",
        "Global average",
        "Mean score",
        "Overall score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "DeepSeek-R1",
      "numericRowCount": 17,
      "observedAtMax": "2025-01-20",
      "observedAtMin": "2025-01-20",
      "rowCount": 17,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 14,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index",
        "epoch-gdpval_external",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-terminalbench_external",
        "epoch-vpct_external",
        "epoch-webdev_arena_external",
        "frontiermath",
        "gpqa-diamond",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1241,
          "metricId": "mean_score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=89;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.000000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://drb.futuresearch.ai",
          "line": 1587,
          "metricId": "Average score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.483",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=20;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 48.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1841,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=133;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "Accuracy mean",
        "Arena Score",
        "Average score",
        "Correct",
        "ECI Score",
        "Mean score",
        "Score (AVG@5)",
        "Win Rate (%)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-opus-4-1-20250805",
      "numericRowCount": 17,
      "observedAtMax": "2026-01-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 17,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 776,
          "metricId": "Challenge score",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.506",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=51;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 50.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 937,
          "metricId": "Average",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.38",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=44;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.38
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2307.09288",
          "line": 1076,
          "metricId": "Score",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.79",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=136;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.79
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "mpt-30b",
      "numericRowCount": 17,
      "observedAtMax": "2023-06-22",
      "observedAtMin": "2023-06-22",
      "rowCount": 17,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 17,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-gso_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 78,
          "metricId": "Percent correct",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "60.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=69;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 423,
          "metricId": "Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0299",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=147;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0299
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 652,
          "metricId": "Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.345",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=162;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.345
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 17
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "mean_score"
      ],
      "modelRef": "o3-mini-2025-01-31_high",
      "numericRowCount": 17,
      "observedAtMax": "2025-01-31",
      "observedAtMin": "2025-01-31",
      "rowCount": 17,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 17
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 127,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1408.2050610664717,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=126",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1408.2050610664717
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 518,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1441.024557752514,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=517",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1441.024557752514
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 901,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1411.4408844300242,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=900",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1411.4408844300242
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 16
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4-1-fast-reasoning",
      "numericRowCount": 16,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 16,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 16
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 760,
          "metricId": "Challenge score",
          "observedAt": "2023-09-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.844",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=35;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 84.39999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 767,
          "metricId": "Challenge score",
          "observedAt": "2023-09-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.844",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=42;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 84.39999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 914,
          "metricId": "Average",
          "observedAt": "2023-09-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.534",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=19;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.534
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 16
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "Qwen-14B",
      "numericRowCount": 16,
      "observedAtMax": "2023-09-28",
      "observedAtMin": "2023-09-28",
      "rowCount": 16,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 16
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 4,
          "metricId": "all",
          "observedAt": "2026-02-23",
          "protocol": {
            "harness": "Famou-Agent 2.0",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "64.44 ± 1.18",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=10;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 64.44
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 3,
          "metricId": "high",
          "observedAt": "2026-02-23",
          "protocol": {
            "harness": "Famou-Agent 2.0",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "42.22 ± 2.22",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=10;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 42.22
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 1,
          "metricId": "lite",
          "observedAt": "2026-02-23",
          "protocol": {
            "harness": "Famou-Agent 2.0",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "80.3 ± 1.52",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=10;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 80.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 16
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "Gemini-3-Pro-Preview",
      "numericRowCount": 16,
      "observedAtMax": "2026-02-23",
      "observedAtMin": "2026-01-25",
      "rowCount": 16,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 16
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "arena-agent",
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2600,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.054175053463510474,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=41",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.054175053463510474
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 122,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1411.7032038135117,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=121",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1411.7032038135117
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 495,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1457.3398122670906,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=494",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1457.3398122670906
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 15
      },
      "metricIds": [
        "arena_score_bt",
        "ips"
      ],
      "modelRef": "Inkling Small",
      "numericRowCount": 15,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-08-21",
      "rowCount": 15,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 15
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 515,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1445.2819034539855,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=514",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1445.2819034539855
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 84,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1428.306758477434,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=83",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1428.306758477434
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 849,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1446.1810691363385,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=848",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1446.1810691363385
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 15
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-large-3",
      "numericRowCount": 15,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-08-21",
      "rowCount": 15,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 15
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 514,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1445.8274492458943,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=513",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1445.8274492458943
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 826,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1461.069977278946,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=825",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1461.069977278946
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 97,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1420.8801416887386,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=96",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1420.8801416887386
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 15
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-medium-3.5",
      "numericRowCount": 15,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 15,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 15
      }
    },
    {
      "benchmarkCount": 15,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-balrog_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 56,
          "metricId": "Percent correct",
          "observedAt": "2024-12-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "28.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=44;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 28.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 884,
          "metricId": "Average progress",
          "observedAt": "2024-10-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.193",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=25;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 19.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1477,
          "metricId": "Accuracy",
          "observedAt": "2024-10-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=116;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 15
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "claude-3-5-haiku-20241022",
      "numericRowCount": 15,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-10-22",
      "rowCount": 15,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 15
      }
    },
    {
      "benchmarkCount": 15,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "hle",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 66,
          "metricId": "Percent correct",
          "observedAt": "2025-02-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "44.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=55;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 469,
          "metricId": "Score",
          "observedAt": "2025-02-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.008",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=193;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.008
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 709,
          "metricId": "Score",
          "observedAt": "2025-02-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.103",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=220;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.103
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 15
      },
      "metricIds": [
        "Accuracy",
        "Correct",
        "ECI Score",
        "Global average",
        "Mean score",
        "Overall score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gpt-4.5-preview-2025-02-27",
      "numericRowCount": 15,
      "observedAtMax": "2025-02-27",
      "observedAtMin": "2025-02-27",
      "rowCount": 15,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 15
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 503,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1453.5218877479294,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=502",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1453.5218877479294
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 82,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1428.581717565256,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=81",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1428.581717565256
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 895,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1414.1987267946342,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=894",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1414.1987267946342
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "chatgpt-4o-latest-20250326",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 166,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1368.9583776352922,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=165",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1368.9583776352922
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 551,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1401.4233325313194,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=550",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1401.4233325313194
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 937,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1374.3831220699017,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=936",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1374.3831220699017
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-2.5-flash-lite-preview-06-17-thinking",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 179,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1358.3312117451696,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=178",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1358.3312117451696
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 587,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1346.568495058235,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=586",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1346.568495058235
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 975,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1322.645938608132,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=974",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1322.645938608132
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-3-27b-it",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 152,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1382.3446885372987,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=151",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1382.3446885372987
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 573,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1372.9117951493995,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=572",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1372.9117951493995
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 919,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1391.0161672411389,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=918",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1391.0161672411389
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4.1-2025-04-14",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 193,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1340.5100502796126,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=192",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1340.5100502796126
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 596,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1327.5003540665066,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=595",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1327.5003540665066
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 945,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1367.5774995141717,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=944",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1367.5774995141717
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4.1-mini-2025-04-14",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 135,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1403.5037822648712,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=134",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1403.5037822648712
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 541,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1419.3150637387032,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=540",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1419.3150637387032
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 913,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1399.6359561545237,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=912",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1399.6359561545237
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-5-chat",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 124,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1409.7013595138203,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=123",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1409.7013595138203
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 527,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1429.828692109579,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=526",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1429.828692109579
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 903,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1408.725761842096,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=902",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1408.725761842096
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4-0709",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 165,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1369.570458620915,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=164",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1369.570458620915
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 572,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1373.0611971611058,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=571",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1373.0611971611058
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 925,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1386.5339071479739,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=924",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1386.5339071479739
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-medium-2505",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 516,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1445.1736204754845,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=515",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1445.1736204754845
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 868,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1434.5766435365413,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=867",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1434.5766435365413
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 89,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1425.0521391623222,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=88",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1425.0521391623222
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-medium-2508",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 196,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1338.6554359274683,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=195",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1338.6554359274683
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 591,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1339.5223691919996,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=590",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1339.5223691919996
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 952,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1362.6903750679166,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=951",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1362.6903750679166
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-small-2506",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 253,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1278.1008841534292,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=252",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1278.1008841534292
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 639,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1246.0448318598328,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=638",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1246.0448318598328
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 986,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1310.0716803060395,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=985",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1310.0716803060395
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-small-3.1-24b-instruct-2503",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 125,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1409.2132117660055,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=124",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1409.2132117660055
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 520,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1436.6822822429997,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=519",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1436.6822822429997
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 904,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1408.2340799928907,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=903",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1408.2340799928907
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o3-2025-04-16",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 186,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1352.4724923203128,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=185",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1352.4724923203128
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 586,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1348.1061015176363,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=585",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1348.1061015176363
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 942,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1368.7264683015808,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=941",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1368.7264683015808
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o4-mini-2025-04-16",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 501,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1453.8033859145703,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=500",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1453.8033859145703
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 858,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1439.8730904782847,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=857",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1439.8730904782847
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 98,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1420.8568461568589,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=97",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1420.8568461568589
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-vl-235b-a22b-instruct",
      "numericRowCount": 14,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 14,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 8,
          "metricId": "Score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.552",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=12;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.552
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2401.04088",
          "line": 803,
          "metricId": "Challenge score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.597",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=78;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.699999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 811,
          "metricId": "Challenge score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.873",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=86;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 87.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Mixtral-8x7B-v0.1",
      "numericRowCount": 14,
      "observedAtMax": "2023-12-11",
      "observedAtMin": "2023-12-11",
      "rowCount": 14,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 3,
          "metricId": "Score",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.558",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=7;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.558
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 806,
          "metricId": "Challenge score",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.916",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=81;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 91.60000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 934,
          "metricId": "Average",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.814",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=41;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.814
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score",
        "mean_score"
      ],
      "modelRef": "Phi-3-medium-128k-instruct",
      "numericRowCount": 14,
      "observedAtMax": "2024-04-23",
      "observedAtMin": "2024-04-23",
      "rowCount": 14,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 14,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vpct_external",
        "frontiermath",
        "gpqa-diamond",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 430,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0236",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=154;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0236
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 639,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4183",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=149;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4183
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1185,
          "metricId": "mean_score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=33;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 20.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 14
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Correct",
        "ECI Score",
        "Mean score",
        "Score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "o4-mini-2025-04-16_medium",
      "numericRowCount": 14,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 14,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 14
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 183,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1354.1216248864175,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=182",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1354.1216248864175
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 567,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1378.7052182074726,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=566",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1378.7052182074726
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 962,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1351.6174558390596,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=961",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1351.6174558390596
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-2.0-flash-001",
      "numericRowCount": 13,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 13,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-search",
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2499,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1204.1098871642791,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=12",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1204.1098871642791
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 41,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1449.9016395463373,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=40",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1449.9016395463373
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 455,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1480.8017952335786,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=454",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1480.8017952335786
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4.20-multi-agent-beta-0309",
      "numericRowCount": 13,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-21",
      "rowCount": 13,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 881,
          "metricId": "Average progress",
          "observedAt": "2024-12-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.23",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=22;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 23.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1483,
          "metricId": "Accuracy",
          "observedAt": "2024-12-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=122;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1983,
          "metricId": "ECI Score",
          "observedAt": "2024-12-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "127.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=288;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 127.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "Global average",
        "Overall score",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "Llama-3.3-70B-Instruct",
      "numericRowCount": 13,
      "observedAtMax": "2024-12-06",
      "observedAtMin": "2024-12-06",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-critpt_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "epoch-spatialviz_bench_external",
        "epoch-weirdml_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 39,
          "metricId": "Percent correct",
          "observedAt": "2025-04-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "15.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=26;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 480,
          "metricId": "Score",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=204;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 721,
          "metricId": "Score",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0438",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=232;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0438
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Mean score",
        "Overall score",
        "Percent correct",
        "Score",
        "Score (AVG@5)"
      ],
      "modelRef": "Llama-4-Maverick-17B-128E-Instruct",
      "numericRowCount": 13,
      "observedAtMax": "2025-04-06",
      "observedAtMin": "2025-04-05",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 7,
          "metricId": "Score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.573",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=11;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.573
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 810,
          "metricId": "Challenge score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.828",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=85;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 82.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1997,
          "metricId": "ECI Score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "116.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=303;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 116.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall score",
        "Score",
        "mean_score"
      ],
      "modelRef": "Meta-Llama-3-8B-Instruct",
      "numericRowCount": 13,
      "observedAtMax": "2024-04-18",
      "observedAtMin": "2024-04-18",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1,
          "metricId": "Score",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.528",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=5;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.528
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 804,
          "metricId": "Challenge score",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.849",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=79;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 84.89999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 932,
          "metricId": "Average",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.717",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=39;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.717
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Global average",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Phi-3-mini-4k-instruct",
      "numericRowCount": 13,
      "observedAtMax": "2024-04-23",
      "observedAtMin": "2024-04-23",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "epoch-wino_grande_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1246,
          "metricId": "mean_score",
          "observedAt": "2024-02-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.05",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=94;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 5.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1707,
          "metricId": "Accuracy",
          "observedAt": "2024-02-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.008199999999999999",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=38;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.8199999999999998
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1847,
          "metricId": "ECI Score",
          "observedAt": "2024-02-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "126.51",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=139;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 126.51
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Global average",
        "Overall score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-3-opus-20240229",
      "numericRowCount": 13,
      "observedAtMax": "2024-02-29",
      "observedAtMin": "2024-02-29",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 13,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-deepresearchbench_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-gso_external",
        "epoch-metr_time_horizons_external",
        "epoch-rli_external",
        "epoch-simplebench_external",
        "epoch-vpct_external",
        "frontiermath",
        "gpqa-diamond",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 30,
          "metricId": "Percent correct",
          "observedAt": "2025-06-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "79.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=17;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://drb.futuresearch.ai",
          "line": 1601,
          "metricId": "Average score",
          "observedAt": "2025-06-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.428",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=34;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1692,
          "metricId": "Accuracy",
          "observedAt": "2025-06-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.055700000000000006",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=23;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 5.57
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "Accuracy",
        "Average score",
        "Correct",
        "ECI Score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gemini-2.5-pro-preview-06-05",
      "numericRowCount": 13,
      "observedAtMax": "2025-06-06",
      "observedAtMin": "2025-06-05",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-blueprint_bench_2_external",
        "epoch-cl_bench_external",
        "epoch-cl_bench_life_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-terminalbench_external",
        "epoch-vending_bench_2_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 121,
          "metricId": "Performance",
          "observedAt": "2026-02-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1150.28",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=35;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1150.28
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 323,
          "metricId": "Score",
          "observedAt": "2026-02-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6514",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=47;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.6514
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 548,
          "metricId": "Score",
          "observedAt": "2026-02-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.895",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=57;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.895
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean",
        "Arena Score",
        "ECI Score",
        "Overall",
        "Overall score",
        "Performance",
        "Score"
      ],
      "modelRef": "grok-4-20",
      "numericRowCount": 13,
      "observedAtMax": "2026-04-02",
      "observedAtMin": "2026-02-17",
      "rowCount": 13,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-lite",
        "swebench-multimodal",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 22,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 64.93,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[21]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 64.93
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 265,
          "metricId": "resolved",
          "observedAt": "2025-06-25",
          "protocol": {
            "checked": false,
            "harness": "ExpeRepair-v1.0",
            "leaderboard_variant": "Lite",
            "scaffold": "ExpeRepair-v1.0",
            "subject_type": "system"
          },
          "rawValue": 60.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[0]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 60.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 267,
          "metricId": "resolved",
          "observedAt": "2025-09-06",
          "protocol": {
            "checked": false,
            "harness": "KGCompass",
            "leaderboard_variant": "Lite",
            "scaffold": "KGCompass",
            "subject_type": "system"
          },
          "rawValue": 58.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[2]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 58.33
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 13
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 4 Sonnet",
      "numericRowCount": 13,
      "observedAtMax": "2025-09-06",
      "observedAtMin": "2025-05-22",
      "rowCount": 13,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 13
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 153,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.4081559205551,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=152",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.4081559205551
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 549,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1404.2075500134918,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=548",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1404.2075500134918
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 940,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1371.6872444481128,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=939",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1371.6872444481128
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-2.5-flash-lite-preview-09-2025-no-thinking",
      "numericRowCount": 12,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 12,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 129,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1406.9268476396717,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=128",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1406.9268476396717
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 504,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1451.3576843349222,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=503",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1451.3576843349222
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 910,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1401.7343569823433,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=909",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1401.7343569823433
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-2.5-flash-preview-09-2025",
      "numericRowCount": 12,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 12,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 491,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1458.8850828367604,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=490",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1458.8850828367604
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 66,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1438.887496531128,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=65",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1438.887496531128
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 848,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1446.213570082081,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=847",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1446.213570082081
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-5.2-chat-latest-20260210",
      "numericRowCount": 12,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 12,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 239,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1287.5554642645209,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=238",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1287.5554642645209
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 624,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1272.324601280256,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=623",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1272.324601280256
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 993,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1302.3937598156624,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=992",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1302.3937598156624
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-4-maverick-17b-128e-instruct",
      "numericRowCount": 12,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 12,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-document",
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-document",
          "benchmarkName": "Arena Document",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2544,
          "metricId": "arena_score_bt",
          "observedAt": "2026-07-30",
          "protocol": {
            "arena_config": "document",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1443.2454382234514,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=document;split=latest;row_idx=23",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1443.2454382234514
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 20,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1473.5252398952205,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=19",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1473.5252398952205
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 431,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1504.9592982097508,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=430",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1504.9592982097508
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "muse-spark",
      "numericRowCount": 12,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-07-30",
      "rowCount": 12,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 748,
          "metricId": "Challenge score",
          "observedAt": "2023-09-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.325",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=23;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 926,
          "metricId": "Average",
          "observedAt": "2023-09-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4156",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=31;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4156
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 943,
          "metricId": "Average",
          "observedAt": "2023-09-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.416",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=50;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.416
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Baichuan-2-7B-Base",
      "numericRowCount": 12,
      "observedAtMax": "2023-09-20",
      "observedAtMin": "2023-09-20",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 1089,
          "metricId": "Score",
          "observedAt": "2022-03-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.837",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=149;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.837
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1112,
          "metricId": "Score",
          "observedAt": "2022-03-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.837",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=176;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.837
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2203.15556",
          "line": 1123,
          "metricId": "Score",
          "observedAt": "2022-03-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.837",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=187;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.837
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "EM",
        "Score"
      ],
      "modelRef": "Chinchilla (70B)",
      "numericRowCount": 12,
      "observedAtMax": "2022-03-29",
      "observedAtMin": "2022-03-29",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 47,
          "metricId": "Percent correct",
          "observedAt": "2025-03-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "55.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=34;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1481,
          "metricId": "Accuracy",
          "observedAt": "2025-03-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=120;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1953,
          "metricId": "ECI Score",
          "observedAt": "2025-03-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "137.08",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=249;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 137.08
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "DeepSeek-V3-0324",
      "numericRowCount": 12,
      "observedAtMax": "2025-03-24",
      "observedAtMin": "2025-03-24",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2112.06905",
          "line": 3180,
          "metricId": "Overall accuracy",
          "observedAt": "2021-12-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.766",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=43;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 76.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2112.06905",
          "line": 3181,
          "metricId": "Overall accuracy",
          "observedAt": "2021-12-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.768",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=44;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 76.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2112.06905",
          "line": 3182,
          "metricId": "Overall accuracy",
          "observedAt": "2021-12-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.772",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=45;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "GLaM (MoE)",
      "numericRowCount": 12,
      "observedAtMax": "2021-12-13",
      "observedAtMin": "2021-12-13",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-bool_q_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-piqa_external",
        "epoch-scicode_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 889,
          "metricId": "Average progress",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.151",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=30;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1064,
          "metricId": "Score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.828",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=124;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.828
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1476,
          "metricId": "Accuracy",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=115;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "Score",
        "mean_score"
      ],
      "modelRef": "Llama-3.1-8B-Instruct",
      "numericRowCount": 12,
      "observedAtMax": "2024-07-23",
      "observedAtMin": "2024-07-23",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 745,
          "metricId": "Challenge score",
          "observedAt": "2023-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.61",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=20;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 924,
          "metricId": "Average",
          "observedAt": "2023-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3368",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=29;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3368
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 940,
          "metricId": "Average",
          "observedAt": "2023-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.337",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=47;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.337
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "chatglm2-6b",
      "numericRowCount": 12,
      "observedAtMax": "2023-06-24",
      "observedAtMin": "2023-06-24",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-frontiermath_tier_4",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1942,
          "metricId": "ECI Score",
          "observedAt": "2024-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "130.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=237;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 130.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2326,
          "metricId": "Overall score",
          "observedAt": "2024-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=43;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2559,
          "metricId": "mean_score",
          "observedAt": "2024-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=66;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "Correct",
        "ECI Score",
        "EM",
        "Overall score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-3-5-sonnet-20240620",
      "numericRowCount": 12,
      "observedAtMax": "2024-06-20",
      "observedAtMin": "2024-06-20",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-algotune_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-simplebench_external",
        "epoch-terminalbench_external",
        "epoch-webdev_arena_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 213,
          "metricId": "Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.34",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=17;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.34
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1685,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.07179100000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=16;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.179100000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2149,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=593;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean",
        "Arena Score",
        "ECI Score",
        "Overall score",
        "Score",
        "Score (AVG@5)"
      ],
      "modelRef": "claude-opus-4-1-20250805_unknown",
      "numericRowCount": 12,
      "observedAtMax": "2025-11-04",
      "observedAtMin": "2025-08-05",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 742,
          "metricId": "Challenge score",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.637",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=17;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 63.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 733,
          "metricId": "Challenge score",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.678",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=8;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 67.80000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1109,
          "metricId": "Score",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.89",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=173;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.89
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "falcon-180B",
      "numericRowCount": 12,
      "observedAtMax": "2023-09-06",
      "observedAtMin": "2023-09-06",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-gsm8k_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-trivia_qa_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 918,
          "metricId": "Average",
          "observedAt": "2023-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.7512",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=23;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.7512
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1165,
          "metricId": "mean_score",
          "observedAt": "2023-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=13;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1749,
          "metricId": "ECI Score",
          "observedAt": "2023-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "121.92",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=35;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 121.92
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "ECI Score",
        "EM",
        "Overall score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4-0613",
      "numericRowCount": 12,
      "observedAtMax": "2023-06-13",
      "observedAtMin": "2023-06-13",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 12,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 40,
          "metricId": "Percent correct",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "32.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=27;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 482,
          "metricId": "Score",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=206;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 723,
          "metricId": "Score",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.035",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=234;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.035
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "gpt-4.1-mini-2025-04-14",
      "numericRowCount": 12,
      "observedAtMax": "2025-04-14",
      "observedAtMin": "2025-04-14",
      "rowCount": 12,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 20,
          "metricId": "all",
          "observedAt": "2026-01-05",
          "protocol": {
            "harness": "PiEvolve<br>(Fractal AI Research)",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "61.33 ± 0.77[^3]",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=14;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 61.33
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 19,
          "metricId": "high",
          "observedAt": "2026-01-05",
          "protocol": {
            "harness": "PiEvolve<br>(Fractal AI Research)",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "40.0 ± 0.00[^3]",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=14;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 40.0
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 17,
          "metricId": "lite",
          "observedAt": "2026-01-05",
          "protocol": {
            "harness": "PiEvolve<br>(Fractal AI Research)",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "80.30 ± 1.52[^3]",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=14;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 80.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "Gemini-3-Pro-Preview[^4]",
      "numericRowCount": 12,
      "observedAtMax": "2026-01-05",
      "observedAtMin": "2025-12-07",
      "rowCount": 12,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 96,
          "metricId": "all",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "8.63 ± 0.54",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=33;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 8.63
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 95,
          "metricId": "high",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "8.15 ± 0.84",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=33;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 8.15
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 93,
          "metricId": "lite",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "18.55 ± 1.26",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=33;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 18.55
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 12
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "gpt-4o-2024-08-06",
      "numericRowCount": 12,
      "observedAtMax": "2024-10-08",
      "observedAtMin": "2024-10-08",
      "rowCount": 12,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 12
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 43,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1447.8483292050098,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=42",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1447.8483292050098
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 448,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1488.2329020092882,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=447",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1488.2329020092882
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 812,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1471.974476060679,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=811",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1471.974476060679
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "dola-seed-2.0-pro",
      "numericRowCount": 11,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 11,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 251,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1280.3266770249836,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=250",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1280.3266770249836
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 635,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1252.1619002705156,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=634",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1252.1619002705156
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1109,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1118.3181942229785,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=108",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1118.3181942229785
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-4-scout-17b-16e-instruct",
      "numericRowCount": 11,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 11,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "arena-text",
        "arena-vision",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 420,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1518.0557525916622,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=419",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1518.0557525916622
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 8,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1487.3676516651792,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=7",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1487.3676516651792
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 778,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1494.157293563849,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=777",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1494.157293563849
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "muse-spark-1.2 (xHigh)",
      "numericRowCount": 11,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 11,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-gso_external",
        "epoch-lech_mazur_writing_external",
        "epoch-simplebench_external",
        "epoch-terminalbench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 26,
          "metricId": "Percent correct",
          "observedAt": "2025-07-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=13;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2018,
          "metricId": "ECI Score",
          "observedAt": "2025-07-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.58",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=336;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.58
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2312,
          "metricId": "Overall score",
          "observedAt": "2025-07-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "60.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=29;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean",
        "ECI Score",
        "Mean score",
        "Overall score",
        "Percent correct",
        "Score (AVG@5)",
        "Score OPT@1"
      ],
      "modelRef": "Kimi-K2-Instruct",
      "numericRowCount": 11,
      "observedAtMax": "2025-11-02",
      "observedAtMin": "2025-07-12",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-spatialviz_bench_external",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 479,
          "metricId": "Score",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=203;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 725,
          "metricId": "Score",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.005",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=236;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.005
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1468,
          "metricId": "Accuracy",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=107;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score",
        "Score",
        "mean_score"
      ],
      "modelRef": "Llama-4-Scout-17B-16E-Instruct",
      "numericRowCount": 11,
      "observedAtMax": "2025-04-05",
      "observedAtMin": "2025-04-05",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 759,
          "metricId": "Challenge score",
          "observedAt": "2023-09-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.753",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=34;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 898,
          "metricId": "Average",
          "observedAt": "2023-09-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.45",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=3;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.45
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 953,
          "metricId": "Average",
          "observedAt": "2023-09-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.45",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=60;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.45
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "Qwen-7B",
      "numericRowCount": 11,
      "observedAtMax": "2023-09-28",
      "observedAtMin": "2023-09-28",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-apex_agents_external",
        "epoch-critpt_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-gso_external",
        "epoch-mindcube_external",
        "epoch-scicode_external",
        "epoch-simplebench_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 264,
          "metricId": "Pass@1 score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.09300000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=50;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1458,
          "metricId": "Accuracy",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00285714285714286",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=97;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.28571428571428603
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1697,
          "metricId": "Accuracy",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.031200000000000002",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=28;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score",
        "Pass@1 score",
        "Score",
        "Score (AVG@5)",
        "Score OPT@1"
      ],
      "modelRef": "claude-sonnet-4-20250514_unknown",
      "numericRowCount": 11,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-the_agent_company_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 470,
          "metricId": "Score",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.008",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=194;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.008
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 882,
          "metricId": "Average progress",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.21",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=23;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1986,
          "metricId": "ECI Score",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "132.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=291;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 132.69
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "% Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gemini-1.5-pro-002",
      "numericRowCount": 11,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-09-24",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-the_agent_company_external",
        "epoch-vpct_external",
        "gpqa-diamond",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 29,
          "metricId": "Percent correct",
          "observedAt": "2025-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "76.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=16;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 76.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1700,
          "metricId": "Accuracy",
          "observedAt": "2025-05-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0236",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=31;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.36
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1947,
          "metricId": "ECI Score",
          "observedAt": "2025-05-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "143.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=242;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 143.04
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "% Score",
        "ACW Avg Score",
        "Accuracy",
        "Correct",
        "ECI Score",
        "Mean score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "gemini-2.5-pro-preview-05-06",
      "numericRowCount": 11,
      "observedAtMax": "2025-05-10",
      "observedAtMin": "2025-05-06",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-apex_agents_external",
        "epoch-critpt_external",
        "epoch-deepresearchbench_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-scicode_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 146,
          "metricId": "Performance",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "797.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=60;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 797.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 261,
          "metricId": "Pass@1 score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.13",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=47;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 13.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1444,
          "metricId": "Accuracy",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0114285714285714",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=83;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.1428571428571401
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Average score",
        "ECI Score",
        "Overall score",
        "Pass@1 score",
        "Performance",
        "Score"
      ],
      "modelRef": "gemini-3.1-flash-lite",
      "numericRowCount": 11,
      "observedAtMax": "2026-03-03",
      "observedAtMin": "2026-03-03",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 65,
          "metricId": "Percent correct",
          "observedAt": "2025-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "4.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=54;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1234,
          "metricId": "mean_score",
          "observedAt": "2025-03-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=82;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1475,
          "metricId": "Accuracy",
          "observedAt": "2025-03-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=114;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemma-3-27b-it",
      "numericRowCount": 11,
      "observedAtMax": "2025-03-15",
      "observedAtMin": "2025-03-12",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 41,
          "metricId": "Percent correct",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "8.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=28;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 478,
          "metricId": "Score",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=202;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 726,
          "metricId": "Score",
          "observedAt": "2025-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=237;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "gpt-4.1-nano-2025-04-14",
      "numericRowCount": 11,
      "observedAtMax": "2025-04-14",
      "observedAtMin": "2025-04-14",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 76,
          "metricId": "Percent correct",
          "observedAt": "2024-12-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "23.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=66;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 23.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1247,
          "metricId": "mean_score",
          "observedAt": "2024-08-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.13",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=95;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 13.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1848,
          "metricId": "ECI Score",
          "observedAt": "2024-08-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "129.05",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=140;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 129.05
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Global average",
        "Percent correct",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4o-2024-08-06",
      "numericRowCount": 11,
      "observedAtMax": "2024-12-30",
      "observedAtMin": "2024-08-06",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-proofbench_external",
        "epoch-scicode_external",
        "epoch-webdev_arena_external",
        "frontiermath",
        "gpqa-diamond",
        "hle",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1401,
          "metricId": "Accuracy",
          "observedAt": "2026-04-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.11330612244898002",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=40;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.330612244898001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1892,
          "metricId": "ECI Score",
          "observedAt": "2026-04-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "152.56",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=185;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 152.56
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2511,
          "metricId": "mean_score",
          "observedAt": "2026-04-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.146",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=18;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "muse-spark",
      "numericRowCount": 11,
      "observedAtMax": "2026-04-08",
      "observedAtMin": "2026-04-08",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-apex_agents_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 73,
          "metricId": "Percent correct",
          "observedAt": "2024-12-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "61.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=63;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 276,
          "metricId": "Pass@1 score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.011000000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=62;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1164,
          "metricId": "mean_score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.15",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=12;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Pass@1 score",
        "Percent correct",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "o1-2024-12-17_high",
      "numericRowCount": 11,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-17",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-vpct_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 664,
          "metricId": "Score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.307",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=175;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.307
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1163,
          "metricId": "mean_score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=11;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 12.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1747,
          "metricId": "ECI Score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.67",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=33;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "ACW Avg Score",
        "Correct",
        "ECI Score",
        "Mean score",
        "Score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "o1-2024-12-17_medium",
      "numericRowCount": 11,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-12-17",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 11,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 74,
          "metricId": "Percent correct",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "53.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=64;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 435,
          "metricId": "Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.020800000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=159;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.020800000000000003
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 681,
          "metricId": "Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2233",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=192;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.2233
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "o3-mini-2025-01-31_medium",
      "numericRowCount": 11,
      "observedAtMax": "2025-01-31",
      "observedAtMin": "2025-01-31",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4,
          "metricId": "Score",
          "observedAt": "2023-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.425",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=8;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.425
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 807,
          "metricId": "Challenge score",
          "observedAt": "2023-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.759",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=82;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 935,
          "metricId": "Average",
          "observedAt": "2023-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.594",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=42;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.594
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "phi-2",
      "numericRowCount": 11,
      "observedAtMax": "2023-12-12",
      "observedAtMin": "2023-12-12",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-the_agent_company_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 888,
          "metricId": "Average progress",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.162",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=29;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 16.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1992,
          "metricId": "ECI Score",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "129.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=298;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 129.07
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2347,
          "metricId": "Overall score",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=64;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 11
      },
      "metricIds": [
        "% Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "Overall score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "qwen2.5-72b-instruct",
      "numericRowCount": 11,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-09-19",
      "rowCount": 11,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 11
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-the_agent_company_external",
        "epoch-weirdml_external",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1991,
          "metricId": "ECI Score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "128.97",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=297;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 128.97
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2317,
          "metricId": "Overall score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=34;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3540,
          "metricId": "mean_score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.49773413897280966",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=73;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 49.77341389728097
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "% Score",
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "Llama-3.1-405B-Instruct",
      "numericRowCount": 10,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-07-23",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 2,
          "metricId": "Score",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.581",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=6;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.581
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 805,
          "metricId": "Challenge score",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.907",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=80;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 90.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 933,
          "metricId": "Average",
          "observedAt": "2024-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.791",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=40;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.791
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Global average",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Phi-3-small-8k-instruct",
      "numericRowCount": 10,
      "observedAtMax": "2024-04-23",
      "observedAtMin": "2024-04-23",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 757,
          "metricId": "Challenge score",
          "observedAt": "2023-07-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.861",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=32;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 86.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 951,
          "metricId": "Average",
          "observedAt": "2023-07-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.693",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=58;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.693
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 1022,
          "metricId": "Score",
          "observedAt": "2023-07-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.894",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=17;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.894
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "StableBeluga2",
      "numericRowCount": 10,
      "observedAtMax": "2023-07-20",
      "observedAtMin": "2023-07-20",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-piqa_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 890,
          "metricId": "Average progress",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.146",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=31;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1974,
          "metricId": "ECI Score",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "130.36",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=279;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 130.36
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-geobench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://geobench.org/",
          "line": 2650,
          "metricId": "ACW Avg Score",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "3934",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "geobench_external.csv:row=8;column=ACW Avg Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 3934.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemini-1.5-flash-002",
      "numericRowCount": 10,
      "observedAtMax": "2024-09-24",
      "observedAtMin": "2024-09-24",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-the_agent_company_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 450,
          "metricId": "Score",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.013000000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=174;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.013000000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1965,
          "metricId": "ECI Score",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=269;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-geobench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://geobench.org/",
          "line": 2653,
          "metricId": "ACW Avg Score",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "3897",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "geobench_external.csv:row=11;column=ACW Avg Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 3897.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "% Score",
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Global average",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemini-2.0-flash-001",
      "numericRowCount": 10,
      "observedAtMax": "2025-02-05",
      "observedAtMin": "2024-12-17",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-balrog_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-simplebench_external",
        "gpqa-diamond",
        "hle",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 49,
          "metricId": "Percent correct",
          "observedAt": "2025-03-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "72.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=36;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 72.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 867,
          "metricId": "Average progress",
          "observedAt": "2025-03-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.433",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=8;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1694,
          "metricId": "Accuracy",
          "observedAt": "2025-03-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0414",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=25;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.14
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gemini-2.5-pro-exp-03-25",
      "numericRowCount": 10,
      "observedAtMax": "2025-03-25",
      "observedAtMin": "2025-03-25",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.08295",
          "line": 816,
          "metricId": "Challenge score",
          "observedAt": "2024-02-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.421",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=91;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2403.08295",
          "line": 900,
          "metricId": "Average",
          "observedAt": "2024-02-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.352",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=5;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.352
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.08295",
          "line": 1029,
          "metricId": "Score",
          "observedAt": "2024-02-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.694",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=24;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.694
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "gemma-2b",
      "numericRowCount": 10,
      "observedAtMax": "2024-02-21",
      "observedAtMin": "2024-02-21",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1260,
          "metricId": "mean_score",
          "observedAt": "2024-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.06",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=108;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1858,
          "metricId": "ECI Score",
          "observedAt": "2024-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "127.61",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=150;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 127.61
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2325,
          "metricId": "Overall score",
          "observedAt": "2024-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=42;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4-turbo-2024-04-09",
      "numericRowCount": 10,
      "observedAtMax": "2024-04-09",
      "observedAtMin": "2024-04-09",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-apex_agents_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-simplebench_external",
        "epoch-surface_evolver_bench_external",
        "epoch-terminalbench_external",
        "epoch-vending_bench_2_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 172,
          "metricId": "Performance",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "575.62",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=86;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 575.62
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 271,
          "metricId": "Pass@1 score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.047",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=57;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2052,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=394;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy mean",
        "ECI Score",
        "Mean score",
        "Pass@1 score",
        "Performance",
        "Score",
        "Score (AVG@5)",
        "average_score"
      ],
      "modelRef": "gpt-oss-120b",
      "numericRowCount": 10,
      "observedAtMax": "2025-11-03",
      "observedAtMin": "2025-08-05",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 747,
          "metricId": "Challenge score",
          "observedAt": "2023-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.817",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=22;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 81.69999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 942,
          "metricId": "Average",
          "observedAt": "2023-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.525",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=49;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.525
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 1012,
          "metricId": "Score",
          "observedAt": "2023-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.875",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=6;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.875
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "internlm-20b",
      "numericRowCount": 10,
      "observedAtMax": "2023-09-18",
      "observedAtMin": "2023-09-18",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-apex_agents_external",
        "epoch-critpt_external",
        "epoch-deepswe_external",
        "epoch-epoch_capabilities_index",
        "epoch-gbaeval_external",
        "epoch-gdp_pdf_external",
        "epoch-proofbench_external",
        "epoch-scicode_external",
        "epoch-vending_bench_2_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 219,
          "metricId": "Pass@1 score",
          "observedAt": "2026-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.419",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=5;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1394,
          "metricId": "Accuracy",
          "observedAt": "2026-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.151",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=33;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepswe_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1628,
          "metricId": "Pass@1",
          "observedAt": "2026-07-09",
          "protocol": {
            "harness": "mini-swe-agent",
            "subject_type": "system"
          },
          "rawValue": "0.5331858407079646",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepswe_external.csv:row=20;column=Pass@1",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.31858407079646
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "GDP.pdf score",
        "Overall score",
        "Pass@1",
        "Pass@1 score",
        "Score"
      ],
      "modelRef": "muse-spark-1.1",
      "numericRowCount": 10,
      "observedAtMax": "2026-07-09",
      "observedAtMin": "2026-07-09",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 699,
          "metricId": "Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.14",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=210;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.14
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1976,
          "metricId": "ECI Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=281;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.64
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3390,
          "metricId": "Mean score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "6.49",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=26;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.49
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Mean score",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "o1-mini-2024-09-12_medium",
      "numericRowCount": 10,
      "observedAtMax": "2024-09-12",
      "observedAtMin": "2024-09-12",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 10,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-algotune_external",
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 85,
          "metricId": "Percent correct",
          "observedAt": "2025-08-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "41.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=76;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 211,
          "metricId": "Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.41",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=15;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.41
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1302,
          "metricId": "mean_score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=150;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 20.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 10
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "openai/gpt-oss-120b_high",
      "numericRowCount": 10,
      "observedAtMax": "2025-08-06",
      "observedAtMin": "2025-08-05",
      "rowCount": 10,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 10
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 230,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1297.7983623673867,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=229",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1297.7983623673867
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 626,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1270.7789376870135,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=625",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1270.7789376870135
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 966,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1342.9913434207974,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=965",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1342.9913434207974
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-5-sonnet-20241022",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 227,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1299.3290373543246,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=226",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1299.3290373543246
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 613,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1298.5430532980047,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=612",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1298.5430532980047
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 968,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1338.6951187083625,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=967",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1338.6951187083625
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-7-sonnet-20250219",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 197,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1338.2013330439527,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=196",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1338.2013330439527
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 590,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1343.0751416399366,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=589",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1343.0751416399366
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 932,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.374555429628,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=931",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.374555429628
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-sonnet-4-20250514",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1234,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1266.5609721938754,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=233",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1266.5609721938754
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1291,
          "metricId": "arena_score_bt",
          "observedAt": "2026-01-09",
          "protocol": {
            "arena_config": "vision",
            "category": "creative_writing",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1336.0758612512736,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=290",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1336.0758612512736
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1377,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "creative_writing_vision",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1235.706988528621,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=376",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1235.706988528621
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ernie-5.0-preview-1220",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 160,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1374.6470054087408,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=159",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1374.6470054087408
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 544,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1416.3957013398583,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=543",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1416.3957013398583
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 918,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1391.6185273264969,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=917",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1391.6185273264969
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4.6v",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 130,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1406.1653917061874,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=129",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1406.1653917061874
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 887,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1420.798358432358,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=886",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1420.798358432358
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1316,
          "metricId": "arena_score_bt",
          "observedAt": "2026-01-09",
          "protocol": {
            "arena_config": "vision",
            "category": "creative_writing",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1175.7064556220362,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=315",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1175.7064556220362
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-vision-1.5-thinking",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 136,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1400.5917237573901,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=135",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1400.5917237573901
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 507,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1450.8023718408713,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=506",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1450.8023718408713
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 879,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1428.1846297117097,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=878",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1428.1846297117097
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-vl-235b-a22b-thinking",
      "numericRowCount": 9,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 9,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 858,
          "metricId": "Challenge score",
          "observedAt": "2023-03-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.324",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=144;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1137,
          "metricId": "Score",
          "observedAt": "2023-03-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.611",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=201;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.611
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2059,
          "metricId": "ECI Score",
          "observedAt": "2023-03-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "82.27",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=407;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 82.27
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Cerebras-GPT-13B",
      "numericRowCount": 9,
      "observedAtMax": "2023-03-20",
      "observedAtMin": "2023-03-20",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 794,
          "metricId": "Challenge score",
          "observedAt": "2024-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.922",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=69;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 92.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2412.19437",
          "line": 905,
          "metricId": "Average",
          "observedAt": "2024-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.788",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=10;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.788
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 964,
          "metricId": "Average",
          "observedAt": "2024-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.788",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=71;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.788
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "DeepSeek-V2",
      "numericRowCount": 9,
      "observedAtMax": "2024-05-07",
      "observedAtMin": "2024-05-07",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 796,
          "metricId": "Challenge score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.953",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=71;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 95.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2412.19437",
          "line": 907,
          "metricId": "Average",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.829",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=12;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.829
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 966,
          "metricId": "Average",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.829",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=73;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.829
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Llama-3.1-405B",
      "numericRowCount": 9,
      "observedAtMax": "2024-07-23",
      "observedAtMin": "2024-07-23",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-the_agent_company_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 879,
          "metricId": "Average progress",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.27899999999999997",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=20;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1994,
          "metricId": "ECI Score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "125.48",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=300;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 125.48
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3541,
          "metricId": "mean_score",
          "observedAt": "2024-07-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.36678625377643503",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=74;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 36.678625377643506
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "% Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "Llama-3.1-70B-Instruct",
      "numericRowCount": 9,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-07-23",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 758,
          "metricId": "Challenge score",
          "observedAt": "2023-11-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.532",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=33;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 897,
          "metricId": "Average",
          "observedAt": "2023-11-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.282",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=2;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.282
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 952,
          "metricId": "Average",
          "observedAt": "2023-11-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.282",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=59;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.282
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "EM",
        "Score"
      ],
      "modelRef": "Qwen-1_8B",
      "numericRowCount": 9,
      "observedAtMax": "2023-11-30",
      "observedAtMin": "2023-11-30",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 795,
          "metricId": "Challenge score",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.945",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=70;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 94.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2412.19437",
          "line": 906,
          "metricId": "Average",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.798",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=11;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.798
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 965,
          "metricId": "Average",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.798",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=72;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.798
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Qwen2.5-72B",
      "numericRowCount": 9,
      "observedAtMax": "2024-09-19",
      "observedAtMin": "2024-09-19",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vending_bench_2_external",
        "epoch-weirdml_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1306,
          "metricId": "mean_score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=154;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 12.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1919,
          "metricId": "ECI Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=212;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.77
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3402,
          "metricId": "Mean score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "8.24",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=38;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.24
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Mean score",
        "Score",
        "mean_score"
      ],
      "modelRef": "Qwen3-235B-A22B-Thinking-2507",
      "numericRowCount": 9,
      "observedAtMax": "2025-07-25",
      "observedAtMin": "2025-07-25",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 974,
          "metricId": "Average",
          "observedAt": "2023-11-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.514",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=82;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.514
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 981,
          "metricId": "Average",
          "observedAt": "2023-11-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.717",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=90;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.717
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2004,
          "metricId": "ECI Score",
          "observedAt": "2023-11-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "117.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=311;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 117.04
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "Yi-34B-Chat",
      "numericRowCount": 9,
      "observedAtMax": "2023-11-22",
      "observedAtMin": "2023-11-22",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 471,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.007000000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=195;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.007000000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 669,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.286",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=180;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.286
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1977,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=282;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "ECI Score",
        "Mean score",
        "Score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_16K",
      "numericRowCount": 9,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-science_qa_external",
        "epoch-weirdml_external",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1996,
          "metricId": "ECI Score",
          "observedAt": "2024-03-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "117.54",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=302;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 117.54
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2362,
          "metricId": "Overall score",
          "observedAt": "2024-03-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "53.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=79;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3537,
          "metricId": "mean_score",
          "observedAt": "2024-03-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1487915407854985",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=70;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.879154078549849
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall score",
        "Score",
        "mean_score"
      ],
      "modelRef": "claude-3-haiku-20240307",
      "numericRowCount": 9,
      "observedAtMax": "2024-03-07",
      "observedAtMin": "2024-03-07",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 857,
          "metricId": "Challenge score",
          "observedAt": "2023-04-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.396",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=143;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1136,
          "metricId": "Score",
          "observedAt": "2023-04-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.563",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=200;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.563
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2061,
          "metricId": "ECI Score",
          "observedAt": "2023-04-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "88.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=412;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 88.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "dolly-v2-12b",
      "numericRowCount": 9,
      "observedAtMax": "2023-04-11",
      "observedAtMin": "2023-04-11",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2065,
          "metricId": "ECI Score",
          "observedAt": "2024-05-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "109.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=416;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 109.07
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 3091,
          "metricId": "EM",
          "observedAt": "2024-05-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5383",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=214;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.83
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 3173,
          "metricId": "Overall accuracy",
          "observedAt": "2024-05-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8207",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=36;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 82.07
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "falcon-11b",
      "numericRowCount": 9,
      "observedAtMax": "2024-05-09",
      "observedAtMin": "2024-05-09",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "gpqa-diamond",
        "hle",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 52,
          "metricId": "Percent correct",
          "observedAt": "2025-01-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "18.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=40;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 18.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1704,
          "metricId": "Accuracy",
          "observedAt": "2025-01-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.011000000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=35;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1971,
          "metricId": "ECI Score",
          "observedAt": "2025-01-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.19",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=276;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.19
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gemini-2.0-flash-thinking-exp-01-21",
      "numericRowCount": 9,
      "observedAtMax": "2025-01-21",
      "observedAtMin": "2025-01-21",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-adversarial_nli_external",
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-adversarial_nli_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 9,
          "metricId": "Score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.581",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "adversarial_nli_external.csv:row=13;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.581
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 812,
          "metricId": "Challenge score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.874",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=87;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 87.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2002,
          "metricId": "ECI Score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.21",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=308;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.21
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Score",
        "mean_score"
      ],
      "modelRef": "gpt-3.5-turbo-1106",
      "numericRowCount": 9,
      "observedAtMax": "2023-11-06",
      "observedAtMin": "2023-11-06",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1978,
          "metricId": "ECI Score",
          "observedAt": "2024-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "130.76",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=283;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 130.76
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3391,
          "metricId": "Mean score",
          "observedAt": "2024-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "6.36",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=27;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.36
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3512,
          "metricId": "mean_score",
          "observedAt": "2024-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6351963746223565",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=45;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 63.51963746223564
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Mean score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "grok-2-1212",
      "numericRowCount": 9,
      "observedAtMax": "2024-12-12",
      "observedAtMin": "2024-12-12",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 46,
          "metricId": "Percent correct",
          "observedAt": "2025-04-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "53.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=33;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 877,
          "metricId": "Average progress",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.295",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=18;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1945,
          "metricId": "ECI Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "139.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=240;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 139.07
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Average progress",
        "ECI Score",
        "Mean score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "grok-3-beta",
      "numericRowCount": 9,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-09",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-blueprint_bench_2_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-gdp_pdf_external",
        "epoch-vending_bench_2_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 132,
          "metricId": "Performance",
          "observedAt": "2026-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "944.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=46;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 944.17
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-blueprint_bench_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1001,
          "metricId": "Score",
          "observedAt": "2026-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "blueprint_bench_2_external.csv:row=18;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1488,
          "metricId": "Accuracy",
          "observedAt": "2026-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=127;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "GDP.pdf score",
        "Overall score",
        "Performance",
        "Score"
      ],
      "modelRef": "grok-4-3",
      "numericRowCount": 9,
      "observedAtMax": "2026-04-17",
      "observedAtMin": "06/06/2026",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-gdp_pdf_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-proofbench_external",
        "epoch-scicode_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1277,
          "metricId": "mean_score",
          "observedAt": "2026-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.25",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=125;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1413,
          "metricId": "Accuracy",
          "observedAt": "2026-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.08",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=52;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1874,
          "metricId": "ECI Score",
          "observedAt": "2026-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "149.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=166;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 149.16
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "GDP.pdf score",
        "Score",
        "mean_score"
      ],
      "modelRef": "grok-4.3_high",
      "numericRowCount": 9,
      "observedAtMax": "2026-04-17",
      "observedAtMin": "2026-04-17",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 746,
          "metricId": "Challenge score",
          "observedAt": "2023-07-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.695",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=21;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 69.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/pdf/2309.16609",
          "line": 941,
          "metricId": "Average",
          "observedAt": "2023-07-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.37",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=48;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.37
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 1011,
          "metricId": "Score",
          "observedAt": "2023-07-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.641",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=5;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.641
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Average",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "internlm-7b",
      "numericRowCount": 9,
      "observedAtMax": "2023-07-05",
      "observedAtMin": "2023-07-05",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiercode_external",
        "epoch-proofbench_external",
        "epoch-scicode_external",
        "epoch-surface_evolver_bench_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 153,
          "metricId": "Performance",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "763.98",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=67;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 763.98
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1485,
          "metricId": "Accuracy",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=124;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2055,
          "metricId": "ECI Score",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.51",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=397;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.51
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "Main score",
        "Mean score",
        "Performance",
        "Score"
      ],
      "modelRef": "mistral-medium-2604",
      "numericRowCount": 9,
      "observedAtMax": "2026-04-28",
      "observedAtMin": "2026-04-28",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 9,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-chess_puzzles",
        "epoch-cl_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vending_bench_2_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 183,
          "metricId": "Performance",
          "observedAt": "2025-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "370.45",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=97;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 370.45
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1307,
          "metricId": "mean_score",
          "observedAt": "2025-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=155;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-cl_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1335,
          "metricId": "Overall",
          "observedAt": "2025-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.145",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "cl_bench_external.csv:row=20;column=Overall",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.145
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "ECI Score",
        "Overall",
        "Performance",
        "Score",
        "mean_score"
      ],
      "modelRef": "qwen3-max-2025-09-23",
      "numericRowCount": 9,
      "observedAtMax": "2025-09-24",
      "observedAtMin": "2025-09-24",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 848,
          "metricId": "Challenge score",
          "observedAt": "2023-06-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.412",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=134;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.199999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1127,
          "metricId": "Score",
          "observedAt": "2023-06-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.743",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=191;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.743
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2087,
          "metricId": "ECI Score",
          "observedAt": "2023-06-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "92.61",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=467;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 92.61
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 9
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "xgen-7b-8k-base",
      "numericRowCount": 9,
      "observedAtMax": "2023-06-27",
      "observedAtMin": "2023-06-27",
      "rowCount": 9,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 9
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 219,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1314.7839468830207,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=218",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1314.7839468830207
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 610,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1306.7420747674682,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=609",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1306.7420747674682
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 954,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1361.595102779148,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=953",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1361.595102779148
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-7-sonnet-20250219-thinking-32k",
      "numericRowCount": 8,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 8,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 202,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1333.6398111949297,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=201",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1333.6398111949297
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 599,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1324.2815081438484,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=598",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1324.2815081438484
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 963,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1349.3130186867502,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=962",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1349.3130186867502
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4.5v",
      "numericRowCount": 8,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 8,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 200,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1335.487563537792,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=199",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1335.487563537792
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 577,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1358.5949339422302,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=576",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1358.5949339422302
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 967,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1340.1333313501063,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=966",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1340.1333313501063
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "step-1o-turbo-202506",
      "numericRowCount": 8,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 8,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 187,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1349.7639409037006,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=186",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1349.7639409037006
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 550,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1401.6501104105437,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=549",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1401.6501104105437
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 949,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1365.880859278052,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=948",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1365.880859278052
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "step-3",
      "numericRowCount": 8,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 8,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiercode_external",
        "epoch-proofbench_external",
        "epoch-simplebench_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 131,
          "metricId": "Performance",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "946.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=45;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 946.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 360,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3652777777777778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=84;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3652777777777778
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 580,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.795",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=89;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.795
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "Main score",
        "Performance",
        "Score",
        "Score (AVG@5)"
      ],
      "modelRef": "Inkling",
      "numericRowCount": 8,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 880,
          "metricId": "Average progress",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.273",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=21;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1712,
          "metricId": "Accuracy",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0038",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=43;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.38
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1984,
          "metricId": "ECI Score",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "125.67",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=289;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 125.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Average progress",
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "Llama-3.2-90B-Vision-Instruct",
      "numericRowCount": 8,
      "observedAtMax": "2024-09-24",
      "observedAtMin": "2024-09-24",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 770,
          "metricId": "Challenge score",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.555",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=45;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.50000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2402.16819",
          "line": 917,
          "metricId": "Average",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.587",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=22;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.587
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2077,
          "metricId": "ECI Score",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "107.21",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=438;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 107.21
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "Nemotron-4 15B",
      "numericRowCount": 8,
      "observedAtMax": "2024-02-26",
      "observedAtMin": "2024-02-26",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2305.10403",
          "line": 762,
          "metricId": "Challenge score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.596",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=37;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.599999999999994
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 730,
          "metricId": "Challenge score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.596",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=5;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.599999999999994
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1106,
          "metricId": "Score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.881",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=170;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.881
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "PaLM 2-S",
      "numericRowCount": 8,
      "observedAtMax": "2023-05-17",
      "observedAtMin": "2023-05-17",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 788,
          "metricId": "Challenge score",
          "observedAt": "2022-04-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.525",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=63;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 1090,
          "metricId": "Score",
          "observedAt": "2022-04-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.848",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=150;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.848
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2302.13971",
          "line": 3028,
          "metricId": "EM",
          "observedAt": "2022-04-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.33",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=135;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 33.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Score"
      ],
      "modelRef": "PaLM 62B",
      "numericRowCount": 8,
      "observedAtMax": "2022-04-04",
      "observedAtMin": "2022-04-04",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 853,
          "metricId": "Challenge score",
          "observedAt": "2023-05-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.391",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=139;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1132,
          "metricId": "Score",
          "observedAt": "2023-05-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.693",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=196;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.693
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 3250,
          "metricId": "Overall accuracy",
          "observedAt": "2023-05-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.703",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=122;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 70.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "RedPajama-INCITE-7B-Base",
      "numericRowCount": 8,
      "observedAtMax": "2023-05-04",
      "observedAtMin": "2023-05-04",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 727,
          "metricId": "Challenge score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.503",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=2;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 50.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 909,
          "metricId": "Average",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.428",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=14;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.428
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2088,
          "metricId": "ECI Score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "104.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=468;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 104.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Yi-6B",
      "numericRowCount": 8,
      "observedAtMax": "2023-11-02",
      "observedAtMin": "2023-11-02",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vpct_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 160,
          "metricId": "Performance",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "674.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=74;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 674.77
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1938,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=233;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.44
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3406,
          "metricId": "Mean score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "8.45",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=42;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.45
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Correct",
        "ECI Score",
        "Mean score",
        "Performance",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-opus-4-1-20250805_16K",
      "numericRowCount": 8,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 400,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0593",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=124;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0593
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 643,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=153;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1951,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=246;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Mean score",
        "Score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "claude-sonnet-4-20250514_16K",
      "numericRowCount": 8,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-gsm8k_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1066,
          "metricId": "Score",
          "observedAt": "2024-05-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.858",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=126;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.858
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1995,
          "metricId": "ECI Score",
          "observedAt": "2024-05-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "122.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=301;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 122.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2361,
          "metricId": "Overall score",
          "observedAt": "2024-05-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "53.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=78;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Overall score",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemini-1.5-flash-001",
      "numericRowCount": 8,
      "observedAtMax": "2024-05-23",
      "observedAtMin": "2024-05-23",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-otis_mock_aime_2024_2025",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 33,
          "metricId": "Percent correct",
          "observedAt": "2025-05-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "44.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=20;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 444,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0169",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=168;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0169
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 655,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3333",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=165;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3333
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20",
      "numericRowCount": 8,
      "observedAtMax": "2025-05-26",
      "observedAtMin": "2025-05-20",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-scicode_external",
        "epoch-surface_evolver_bench_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 135,
          "metricId": "Performance",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "925.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=49;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 925.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1442,
          "metricId": "Accuracy",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0142857142857143",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=81;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.42857142857143
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1893,
          "metricId": "ECI Score",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=186;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "Mean score",
        "Performance",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemma-4-31b-it",
      "numericRowCount": 8,
      "observedAtMax": "2026-04-02",
      "observedAtMin": "2026-04-02",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-apex_agents_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-scicode_external",
        "epoch-terminalbench_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 186,
          "metricId": "Performance",
          "observedAt": "2025-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "340.82",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=100;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 340.82
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 273,
          "metricId": "Pass@1 score",
          "observedAt": "2025-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=59;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1447,
          "metricId": "Accuracy",
          "observedAt": "2025-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0114285714285714",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=86;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.1428571428571401
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean",
        "Arena Score",
        "ECI Score",
        "Pass@1 score",
        "Performance",
        "Score"
      ],
      "modelRef": "glm-4.6",
      "numericRowCount": 8,
      "observedAtMax": "2026-01-05",
      "observedAtMin": "2025-09-30",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1206,
          "metricId": "mean_score",
          "observedAt": "2024-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=54;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1784,
          "metricId": "ECI Score",
          "observedAt": "2024-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "114.33",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=76;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 114.33
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2365,
          "metricId": "Overall score",
          "observedAt": "2024-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "50.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=82;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 50.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "gpt-3.5-turbo-0125",
      "numericRowCount": 8,
      "observedAtMax": "2024-01-25",
      "observedAtMin": "2024-01-25",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-common_sense_qa_2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 919,
          "metricId": "Average",
          "observedAt": "2023-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6159",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=24;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.6159
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1036,
          "metricId": "Score",
          "observedAt": "2023-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.87",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=32;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.87
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-common_sense_qa_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2309.13165v1",
          "line": 1361,
          "metricId": "Score",
          "observedAt": "2023-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.57",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "common_sense_qa_2_external.csv:row=10;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.57
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "gpt-3.5-turbo-0613",
      "numericRowCount": 8,
      "observedAtMax": "2023-06-13",
      "observedAtMin": "2023-06-13",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1935,
          "metricId": "ECI Score",
          "observedAt": "2023-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "126.19",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=228;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 126.19
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2303.08774",
          "line": 3092,
          "metricId": "EM",
          "observedAt": "2023-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.92",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=215;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 92.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2303.08774",
          "line": 3197,
          "metricId": "Overall accuracy",
          "observedAt": "2023-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.953",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=60;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 95.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4-0314",
      "numericRowCount": 8,
      "observedAtMax": "2023-03-14",
      "observedAtMin": "2023-03-14",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-science_qa_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 874,
          "metricId": "Average progress",
          "observedAt": "2024-05-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.32299999999999995",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=15;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1993,
          "metricId": "ECI Score",
          "observedAt": "2024-05-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "128.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=299;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 128.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2344,
          "metricId": "Overall score",
          "observedAt": "2024-05-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=61;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Average progress",
        "ECI Score",
        "EM",
        "Overall score",
        "Score",
        "mean_score"
      ],
      "modelRef": "gpt-4o-2024-05-13",
      "numericRowCount": 8,
      "observedAtMax": "2024-05-13",
      "observedAtMin": "2024-05-13",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1284,
          "metricId": "mean_score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.54",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=132;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1888,
          "metricId": "ECI Score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "158.67",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=181;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 158.67
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2507,
          "metricId": "mean_score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.354",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=14;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gpt-5.5-pre-release_xhigh",
      "numericRowCount": 8,
      "observedAtMax": "2026-04-23",
      "observedAtMin": "2026-04-23",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 856,
          "metricId": "Challenge score",
          "observedAt": "2021-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.363",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=142;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 36.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1135,
          "metricId": "Score",
          "observedAt": "2021-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.654",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=199;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.654
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 3198,
          "metricId": "Overall accuracy",
          "observedAt": "2021-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.662",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=62;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 66.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "gpt-j-6b",
      "numericRowCount": 8,
      "observedAtMax": "2021-08-05",
      "observedAtMin": "2021-08-05",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-metr_time_horizons_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 839,
          "metricId": "Challenge score",
          "observedAt": "2019-11-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.25",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=116;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 1103,
          "metricId": "Score",
          "observedAt": "2019-11-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.618",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=165;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.618
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 3201,
          "metricId": "Overall accuracy",
          "observedAt": "2019-11-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=66;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "Overall accuracy",
        "Score",
        "average_score"
      ],
      "modelRef": "gpt2-xl",
      "numericRowCount": 8,
      "observedAtMax": "2019-11-05",
      "observedAtMin": "2019-11-05",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-weirdml_external",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 44,
          "metricId": "Percent correct",
          "observedAt": "2025-04-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "49.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=31;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 49.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1927,
          "metricId": "ECI Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=220;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.04
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3492,
          "metricId": "mean_score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8806646525679759",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=25;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 88.06646525679758
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "grok-3-mini-beta_high",
      "numericRowCount": 8,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-09",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 24,
          "metricId": "Percent correct",
          "observedAt": "2025-04-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "34.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=11;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 34.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 692,
          "metricId": "Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.165",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=203;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.165
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1958,
          "metricId": "ECI Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=255;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.04
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "Mean score",
        "Percent correct",
        "Score",
        "mean_score"
      ],
      "modelRef": "grok-3-mini-beta_low",
      "numericRowCount": 8,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-09",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-mystery_game_puzzles",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 356,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.40138888888888885",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=80;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.40138888888888885
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 574,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=83;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.84
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1157,
          "metricId": "mean_score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.18",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=5;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 18.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "inkling-small_xhigh",
      "numericRowCount": 8,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-apex_agents_external",
        "epoch-cl_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-metr_time_horizons_external",
        "epoch-terminalbench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 170,
          "metricId": "Performance",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "597.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=84;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 597.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 272,
          "metricId": "Pass@1 score",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.040999999999999995",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=58;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-cl_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1338,
          "metricId": "Overall",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.119",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "cl_bench_external.csv:row=23;column=Overall",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.119
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean",
        "ECI Score",
        "Overall",
        "Pass@1 score",
        "Performance",
        "average_score"
      ],
      "modelRef": "kimi-k2-thinking",
      "numericRowCount": 8,
      "observedAtMax": "2025-11-11",
      "observedAtMin": "2025-11-06",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1999,
          "metricId": "ECI Score",
          "observedAt": "2024-07-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "127.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=305;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 127.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2350,
          "metricId": "Overall score",
          "observedAt": "2024-07-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=67;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3388,
          "metricId": "Mean score",
          "observedAt": "2024-07-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "6.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=24;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Mean score",
        "Overall score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "mistral-large-2407",
      "numericRowCount": 8,
      "observedAtMax": "2024-07-24",
      "observedAtMin": "2024-07-24",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-simplebench_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 688,
          "metricId": "Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.18",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=199;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.18
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1972,
          "metricId": "ECI Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=277;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.77
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3574,
          "metricId": "mean_score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8164652567975831",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=107;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 81.64652567975831
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score",
        "Score (AVG@5)",
        "average_score",
        "mean_score"
      ],
      "modelRef": "o1-preview-2024-09-12",
      "numericRowCount": 8,
      "observedAtMax": "2024-09-12",
      "observedAtMin": "2024-09-12",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-gso_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 481,
          "metricId": "Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=205;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 697,
          "metricId": "Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.145",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=208;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.145
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1249,
          "metricId": "mean_score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.06",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=97;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "Global average",
        "Score",
        "Score OPT@1",
        "mean_score"
      ],
      "modelRef": "o3-mini-2025-01-31_low",
      "numericRowCount": 8,
      "observedAtMax": "2025-01-31",
      "observedAtMin": "2025-01-31",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 445,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0167",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=169;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0167
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 683,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2133",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=194;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.2133
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1268,
          "metricId": "mean_score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.14",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=116;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.000000000000002
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "o4-mini-2025-04-16_low",
      "numericRowCount": 8,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 852,
          "metricId": "Challenge score",
          "observedAt": "2023-06-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.387",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=138;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 38.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1131,
          "metricId": "Score",
          "observedAt": "2023-06-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.706",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=195;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.706
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 3227,
          "metricId": "Overall accuracy",
          "observedAt": "2023-06-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.718",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=99;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 71.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "open_llama_7b",
      "numericRowCount": 8,
      "observedAtMax": "2023-06-07",
      "observedAtMin": "2023-06-07",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 855,
          "metricId": "Challenge score",
          "observedAt": "2022-05-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.358",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=141;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1134,
          "metricId": "Score",
          "observedAt": "2022-05-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.65",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=198;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.65
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 3231,
          "metricId": "Overall accuracy",
          "observedAt": "2022-05-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.699",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=103;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 69.89999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "opt-13b",
      "numericRowCount": 8,
      "observedAtMax": "2022-05-11",
      "observedAtMin": "2022-05-11",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 892,
          "metricId": "Average progress",
          "observedAt": "2024-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.11599999999999999",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=33;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1998,
          "metricId": "ECI Score",
          "observedAt": "2024-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "131.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=304;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 131.04
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3392,
          "metricId": "Mean score",
          "observedAt": "2024-12-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "6.26",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=28;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Average progress",
        "ECI Score",
        "EM",
        "Global average",
        "Mean score",
        "mean_score"
      ],
      "modelRef": "phi-4",
      "numericRowCount": 8,
      "observedAtMax": "2024-12-12",
      "observedAtMin": "2024-12-12",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1032,
          "metricId": "Score",
          "observedAt": "2022-11-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.881",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=28;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.881
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 3003,
          "metricId": "EM",
          "observedAt": "2022-11-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.571",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=107;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.099999999999994
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2309.16609",
          "line": 3040,
          "metricId": "EM",
          "observedAt": "2022-11-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.782",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=150;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 78.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "text-davinci-003",
      "numericRowCount": 8,
      "observedAtMax": "2022-11-28",
      "observedAtMin": "2022-11-28",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 8,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 833,
          "metricId": "Challenge score",
          "observedAt": "2023-04-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.432",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=108;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 929,
          "metricId": "Average",
          "observedAt": "2023-04-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4304",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=34;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4304
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 1096,
          "metricId": "Score",
          "observedAt": "2023-04-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.835",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=157;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.835
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "vicuna-13b-v1.1",
      "numericRowCount": 8,
      "observedAtMax": "2023-04-12",
      "observedAtMin": "2023-04-12",
      "rowCount": 8,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 64,
          "metricId": "all",
          "observedAt": "2025-09-12",
          "protocol": {
            "harness": "InternAgent",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "36.44 ± 1.18",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=25;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 36.44
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 63,
          "metricId": "high",
          "observedAt": "2025-09-12",
          "protocol": {
            "harness": "InternAgent",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "24.44 ± 2.22",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=25;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 24.44
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 61,
          "metricId": "lite",
          "observedAt": "2025-09-12",
          "protocol": {
            "harness": "InternAgent",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "62.12 ± 3.03",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=25;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 62.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "deepseek-r1",
      "numericRowCount": 8,
      "observedAtMax": "2025-09-12",
      "observedAtMin": "2025-06-17",
      "rowCount": 8,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 88,
          "metricId": "all",
          "observedAt": "2025-05-14",
          "protocol": {
            "harness": "R&D-Agent",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "22.40 ± 0.50",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=31;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 22.4
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 87,
          "metricId": "high",
          "observedAt": "2025-05-14",
          "protocol": {
            "harness": "R&D-Agent",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "18.67 ± 1.33",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=31;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 18.67
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 85,
          "metricId": "lite",
          "observedAt": "2025-05-14",
          "protocol": {
            "harness": "R&D-Agent",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "48.18 ± 1.11",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=31;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 48.18
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "o1-preview",
      "numericRowCount": 8,
      "observedAtMax": "2025-05-14",
      "observedAtMin": "2024-10-08",
      "rowCount": 8,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-lite",
        "swebench-multimodal",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 35,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 52.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[34]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 52.8
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 278,
          "metricId": "resolved",
          "observedAt": "2025-02-26",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Lite",
            "scaffold": "SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 48.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[13]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 48.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 354,
          "metricId": "resolved",
          "observedAt": "2025-05-09",
          "protocol": {
            "checked": true,
            "harness": "OpenHands-Versa",
            "leaderboard_variant": "Multimodal",
            "scaffold": "OpenHands-Versa",
            "subject_type": "system"
          },
          "rawValue": 31.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[5]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 31.33
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 8
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 3.7 Sonnet",
      "numericRowCount": 8,
      "observedAtMax": "2025-07-20",
      "observedAtMin": "2025-02-24",
      "rowCount": 8,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 8
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 61,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond_high_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=61;row=60",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 80.9
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5d/3e/5d3e8832-c770-4796-957b-dfaefbd3bb2a.json",
          "line": 62,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.8080808081,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=62;row=61",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 80.8080808081
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 66,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond_high.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.1,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=66;row=65",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 80.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-120b",
      "numericRowCount": 7,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 7,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 101,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond_low.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 56.8,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=101;row=100",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 56.8
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 81,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond_high_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74.2,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=81;row=80",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 74.2
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 88,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond_high.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=88;row=87",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 71.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-20b",
      "numericRowCount": 7,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 7,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 271,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1255.253242164797,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=270",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1255.253242164797
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 657,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1224.2240616049103,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=656",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1224.2240616049103
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1112,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1093.2236812894607,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=111",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1093.2236812894607
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-5-haiku-20241022",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 189,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1348.6742603266173,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=188",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1348.6742603266173
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 584,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1352.7738166966124,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=583",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1352.7738166966124
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 896,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1413.1041538103273,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=895",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1413.1041538103273
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-sonnet-4-20250514-thinking-32k",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 234,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1291.5699541226504,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=233",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1291.5699541226504
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 597,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1325.377096151695,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=596",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1325.377096151695
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 983,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1312.7129718333701,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=982",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1312.7129718333701
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "granite-4.1-8b",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 133,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1404.7704693835374,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=132",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1404.7704693835374
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 517,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1442.053592522068,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=516",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1442.053592522068
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 876,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1428.6430538883317,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=875",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1428.6430538883317
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-hy3-preview",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 265,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1263.1955058975161,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=264",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1263.1955058975161
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 621,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1275.5220259120556,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=620",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1275.5220259120556
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 992,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1306.0346079246,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=991",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1306.0346079246
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-large-vision",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 426,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1511.3938042848351,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=425",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1511.3938042848351
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 60,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1441.1730149730124,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=59",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1441.1730149730124
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 819,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1466.8554610217427,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=818",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1466.8554610217427
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hy3",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 442,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1494.8046177188173,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=441",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1494.8046177188173
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 63,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1439.2420048228466,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=62",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1439.2420048228466
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 825,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1461.0879978543724,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=824",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1461.0879978543724
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "inkling",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 117,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1414.2663536929415,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=116",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1414.2663536929415
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 500,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1454.311441418989,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=499",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1454.311441418989
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 837,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1453.3129586493396,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=836",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1453.3129586493396
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "kimi-k2-thinking-turbo",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 147,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1391.0615144949397,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=146",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1391.0615144949397
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 524,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1432.08742133001,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=523",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1432.08742133001
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 888,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1419.9270022363016,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=887",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1419.9270022363016
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "muse-glimmer",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 192,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1341.8837256148295,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=191",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1341.8837256148295
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 574,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1372.0096985482182,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=573",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1372.0096985482182
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 958,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1358.4308041436384,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=957",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1358.4308041436384
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "trinity-large-thinking",
      "numericRowCount": 7,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 7,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2412.01152",
          "line": 860,
          "metricId": "Challenge score",
          "observedAt": "2024-11-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5452",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=146;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.52
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://arxiv.org/abs/2412.01152",
          "line": 982,
          "metricId": "Average",
          "observedAt": "2024-11-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3485",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=91;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3485
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2069,
          "metricId": "ECI Score",
          "observedAt": "2024-11-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "100.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=427;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 100.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "Average",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "INTELLECT-1-Instruct",
      "numericRowCount": 7,
      "observedAtMax": "2024-11-29",
      "observedAtMin": "2024-11-29",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1235,
          "metricId": "mean_score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.21",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=83;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1416,
          "metricId": "Accuracy",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0542857142857143",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=55;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 5.42857142857143
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1812,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "148.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=104;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 148.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "Inkling_xhigh",
      "numericRowCount": 7,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 968,
          "metricId": "Average",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.329",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=75;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.329
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 975,
          "metricId": "Average",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.582",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=83;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.582
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2228,
          "metricId": "ECI Score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "105.62",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=763;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 105.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM"
      ],
      "modelRef": "Llama-2-13b-chat",
      "numericRowCount": 7,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 969,
          "metricId": "Average",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.424",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=76;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.424
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 976,
          "metricId": "Average",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.585",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=84;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.585
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2229,
          "metricId": "ECI Score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "113.57",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=764;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 113.57
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM"
      ],
      "modelRef": "Llama-2-70b-chat",
      "numericRowCount": 7,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 196,
          "metricId": "Performance",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "172.97",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=110;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 172.97
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1959,
          "metricId": "ECI Score",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "133.03",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=256;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 133.03
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2348,
          "metricId": "Overall score",
          "observedAt": "2025-04-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=65;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "ECI Score",
        "Overall score",
        "Performance",
        "mean_score"
      ],
      "modelRef": "Llama-4-Maverick-17B-128E-Instruct-FP8",
      "numericRowCount": 7,
      "observedAtMax": "2025-04-05",
      "observedAtMin": "2025-04-05",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2305.10403",
          "line": 763,
          "metricId": "Challenge score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.649",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=38;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 64.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 731,
          "metricId": "Challenge score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.649",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=6;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 64.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1107,
          "metricId": "Score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.886",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=171;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.886
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "PaLM 2-M",
      "numericRowCount": 7,
      "observedAtMax": "2023-05-17",
      "observedAtMin": "2023-05-17",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 971,
          "metricId": "Average",
          "observedAt": "2023-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.497",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=78;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.497
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 978,
          "metricId": "Average",
          "observedAt": "2023-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.55",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=86;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.55
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2230,
          "metricId": "ECI Score",
          "observedAt": "2023-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "112.62",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=766;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 112.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM"
      ],
      "modelRef": "Qwen-14B-Chat",
      "numericRowCount": 7,
      "observedAtMax": "2023-09-24",
      "observedAtMin": "2023-09-24",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 832,
          "metricId": "Challenge score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.705",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=107;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 70.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2085,
          "metricId": "ECI Score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "119.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=458;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 119.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3083,
          "metricId": "EM",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.911",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=206;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 91.10000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Qwen2.5-Coder-32B",
      "numericRowCount": 7,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 973,
          "metricId": "Average",
          "observedAt": "2023-11-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.397",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=81;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.397
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 980,
          "metricId": "Average",
          "observedAt": "2023-11-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.472",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=89;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.472
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2231,
          "metricId": "ECI Score",
          "observedAt": "2023-11-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "104.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=768;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 104.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM"
      ],
      "modelRef": "Yi-6B-Chat",
      "numericRowCount": 7,
      "observedAtMax": "2023-11-22",
      "observedAtMin": "2023-11-22",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-gso_external",
        "epoch-simplebench_external",
        "epoch-spatialviz_bench_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1693,
          "metricId": "Accuracy",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.042300000000000004",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=24;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.23
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2020,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=339;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2288,
          "metricId": "Overall score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "61.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=5;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score",
        "Score (AVG@5)",
        "Score OPT@1"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_unknown",
      "numericRowCount": 7,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-weirdml_external",
        "epoch-wino_grande_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1988,
          "metricId": "ECI Score",
          "observedAt": "2024-02-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "119.83",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=294;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 119.83
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3538,
          "metricId": "mean_score",
          "observedAt": "2024-02-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.18174093655589124",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=71;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 18.174093655589125
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3657,
          "metricId": "EM",
          "observedAt": "2024-02-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.759",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=26;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "claude-3-sonnet-20240229",
      "numericRowCount": 7,
      "observedAtMax": "2024-02-29",
      "observedAtMin": "2024-02-29",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-trivia_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.anthropic.com/news/releasing-claude-instant-1-2",
          "line": 846,
          "metricId": "Challenge score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.857",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=130;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 85.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2209,
          "metricId": "ECI Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "121.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=676;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 121.07
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.anthropic.com/news/releasing-claude-instant-1-2",
          "line": 3094,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.809",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=218;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 80.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "claude-instant-1.1",
      "numericRowCount": 7,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-ale_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vpct_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 35,
          "metricId": "Percent correct",
          "observedAt": "2025-05-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "61.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=22;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 163,
          "metricId": "Performance",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "655.35",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=77;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 655.35
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1949,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=244;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "ACW Avg Score",
        "Correct",
        "ECI Score",
        "Percent correct",
        "Performance",
        "mean_score"
      ],
      "modelRef": "claude-sonnet-4-20250514_32K",
      "numericRowCount": 7,
      "observedAtMax": "2025-05-24",
      "observedAtMin": "2025-05-22",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf",
          "line": 984,
          "metricId": "Average",
          "observedAt": "2024-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.892",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=93;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.892
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1982,
          "metricId": "ECI Score",
          "observedAt": "2024-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "127.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=287;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 127.17
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2337,
          "metricId": "Overall score",
          "observedAt": "2024-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "58.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=54;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 58.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "gemini-1.5-pro-001",
      "numericRowCount": 7,
      "observedAtMax": "2024-05-14",
      "observedAtMin": "2024-05-14",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-geobench_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vpct_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 43,
          "metricId": "Percent correct",
          "observedAt": "2025-04-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "47.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=30;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 47.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1956,
          "metricId": "ECI Score",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=252;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.84
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2307,
          "metricId": "Overall score",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "60.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=24;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "ACW Avg Score",
        "Accuracy",
        "Correct",
        "ECI Score",
        "Overall score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "gemini-2.5-flash-preview-04-17",
      "numericRowCount": 7,
      "observedAtMax": "2025-04-20",
      "observedAtMin": "2025-04-17",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-terminalbench_external",
        "epoch-the_agent_company_external",
        "epoch-vpct_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2021,
          "metricId": "ECI Score",
          "observedAt": "2025-09-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "143.38",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=341;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 143.38
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 5021,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.171",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=177;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 17.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 5023,
          "metricId": "Accuracy mean",
          "observedAt": "2025-10-31",
          "protocol": {
            "harness": "Terminus 2",
            "subject_type": "system"
          },
          "rawValue": "0.169",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=179;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 16.900000000000002
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "% Score",
        "Accuracy mean",
        "Correct",
        "ECI Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-09-2025",
      "numericRowCount": 7,
      "observedAtMax": "2025-11-04",
      "observedAtMin": "2025-09-25",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-simplebench_external",
        "epoch-vpct_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 80,
          "metricId": "Percent correct",
          "observedAt": "2025-04-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "72.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=71;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 72.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 657,
          "metricId": "Score",
          "observedAt": "2025-03-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.33",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=168;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.33
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1954,
          "metricId": "ECI Score",
          "observedAt": "2025-03-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.63",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=250;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.63
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Correct",
        "ECI Score",
        "Overall score",
        "Percent correct",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gemini-2.5-pro-preview-03-25",
      "numericRowCount": 7,
      "observedAtMax": "2025-04-12",
      "observedAtMin": "2025-03-31",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-terminalbench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 4949,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.413",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=104;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4950,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.413",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=105;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 4928,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-04",
          "protocol": {
            "harness": "Codex CLI",
            "subject_type": "system"
          },
          "rawValue": "0.443",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=83;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean"
      ],
      "modelRef": "gpt-5-codex",
      "numericRowCount": 7,
      "observedAtMax": "2025-11-04",
      "observedAtMin": "2025-09-15",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1283,
          "metricId": "mean_score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=131;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 64.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1887,
          "metricId": "ECI Score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "161.71",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=180;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 161.71
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2505,
          "metricId": "mean_score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.396",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=12;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gpt-5.5-pro-pre-release_xhigh",
      "numericRowCount": 7,
      "observedAtMax": "2026-04-23",
      "observedAtMin": "2026-04-23",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-proofbench_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1267,
          "metricId": "mean_score",
          "observedAt": "2026-02-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.24",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=115;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 24.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1865,
          "metricId": "ECI Score",
          "observedAt": "2026-02-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "152.09",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=157;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 152.09
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2306,
          "metricId": "Overall score",
          "observedAt": "2026-02-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "60.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=23;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "grok-4.20-0309-reasoning",
      "numericRowCount": 7,
      "observedAtMax": "2026-02-17",
      "observedAtMin": "2026-02-17",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-terminalbench_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 171,
          "metricId": "Performance",
          "observedAt": "2025-08-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "587.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=85;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 587.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 4999,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.258",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=155;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5000,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.258",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=156;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "Accuracy mean",
        "Arena Score",
        "Performance"
      ],
      "modelRef": "grok-code-fast-1",
      "numericRowCount": 7,
      "observedAtMax": "2026-01-05",
      "observedAtMin": "2025-08-28",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1979,
          "metricId": "ECI Score",
          "observedAt": "2024-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "128.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=284;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 128.77
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2354,
          "metricId": "Overall score",
          "observedAt": "2024-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "56.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=71;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3510,
          "metricId": "mean_score",
          "observedAt": "2024-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5028323262839879",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=43;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 50.283232628398785
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "ECI Score",
        "Global average",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "mistral-large-2411",
      "numericRowCount": 7,
      "observedAtMax": "2024-11-18",
      "observedAtMin": "2024-11-18",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1955,
          "metricId": "ECI Score",
          "observedAt": "2025-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.33",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=251;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.33
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3414,
          "metricId": "Mean score",
          "observedAt": "2025-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "7.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=50;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3485,
          "metricId": "mean_score",
          "observedAt": "2025-05-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8162764350453172",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=18;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 81.62764350453172
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Mean score",
        "mean_score"
      ],
      "modelRef": "mistral-medium-2505",
      "numericRowCount": 7,
      "observedAtMax": "2025-05-07",
      "observedAtMin": "2025-05-07",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1478,
          "metricId": "Accuracy",
          "observedAt": "2025-03-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=117;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1963,
          "metricId": "ECI Score",
          "observedAt": "2025-03-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "127.71",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=264;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 127.71
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3500,
          "metricId": "mean_score",
          "observedAt": "2025-03-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4677114803625378",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=33;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.77114803625378
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Score",
        "mean_score"
      ],
      "modelRef": "mistral-small-2503",
      "numericRowCount": 7,
      "observedAtMax": "2025-03-17",
      "observedAtMin": "2025-03-17",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 437,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0199",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=161;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0199
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 640,
          "metricId": "Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.415",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=150;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.415
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1248,
          "metricId": "mean_score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.27",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=96;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "o3-2025-04-16_low",
      "numericRowCount": 7,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-open_book_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 840,
          "metricId": "Challenge score",
          "observedAt": "2023-09-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.444",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=119;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 1104,
          "metricId": "Score",
          "observedAt": "2023-09-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.758",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=168;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.758
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2078,
          "metricId": "ECI Score",
          "observedAt": "2023-09-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "90.72",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=448;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 90.72
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "phi-1_5",
      "numericRowCount": 7,
      "observedAtMax": "2023-09-11",
      "observedAtMin": "2023-09-11",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 7,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external",
        "epoch-the_agent_company_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2005,
          "metricId": "ECI Score",
          "observedAt": "2024-06-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "125.34",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=312;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 125.34
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3555,
          "metricId": "mean_score",
          "observedAt": "2024-06-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.39067220543806647",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=88;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.06722054380665
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "line": 3623,
          "metricId": "average_score",
          "observedAt": "2024-06-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.298999",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=48;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.298999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "% Score",
        "Accuracy",
        "ECI Score",
        "EM",
        "average_score",
        "mean_score"
      ],
      "modelRef": "qwen2-72b-instruct",
      "numericRowCount": 7,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-06-07",
      "rowCount": 7,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 319,
          "metricId": "resolved",
          "observedAt": "2024-06-04",
          "protocol": {
            "checked": false,
            "harness": "CodeR",
            "leaderboard_variant": "Lite",
            "scaffold": "CodeR",
            "subject_type": "system"
          },
          "rawValue": 28.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[54]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 28.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 340,
          "metricId": "resolved",
          "observedAt": "2024-04-02",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Lite",
            "scaffold": "SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 18.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[75]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 18.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 345,
          "metricId": "resolved",
          "observedAt": "2024-04-02",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Lite",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 2.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[80]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 2.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 7
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4 (1106)",
      "numericRowCount": 7,
      "observedAtMax": "2024-06-04",
      "observedAtMin": "2024-04-02",
      "rowCount": 7,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 7
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 65,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_high_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 19,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=65;row=64",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 19.0
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 72,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_high.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 14.9,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=72;row=71",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 14.9
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 77,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_medium_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 11.3,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=77;row=76",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 11.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-120b",
      "numericRowCount": 6,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 6,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 71,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_high_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 17.3,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=71;row=70",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 17.3
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 78,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_high.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 10.9,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=78;row=77",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 10.9
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 82,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_medium_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 8.8,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=82;row=81",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 8.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-20b",
      "numericRowCount": 6,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 6,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 249,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1281.047484206804,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=248",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1281.047484206804
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 628,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1263.719641839382,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=627",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1263.719641839382
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 990,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1307.0316861943427,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=989",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1307.0316861943427
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-5-sonnet-20240620",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 109,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1417.5812835340903,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=108",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1417.5812835340903
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 534,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1424.5881690490826,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=533",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1424.5881690490826
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 808,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1473.4097228366018,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=807",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1473.4097228366018
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-opus-4-1-20250805",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 215,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1319.2490899193742,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=214",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1319.2490899193742
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 594,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1330.3852521839394,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=593",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1330.3852521839394
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 997,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1294.8480913558808,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=996",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1294.8480913558808
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-1.5-pro-002",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 207,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1329.6341887862836,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=206",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1329.6341887862836
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 592,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1338.217264883037,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=591",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1338.217264883037
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 976,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1322.485919185502,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=975",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1322.485919185502
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-2.0-flash-lite-preview-02-05",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 438,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1499.5145407918126,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=437",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1499.5145407918126
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 62,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1439.7714291618945,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=61",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1439.7714291618945
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 845,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1448.638421427936,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=844",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1448.638421427936
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4.6",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 110,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1417.4235653366281,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=109",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1417.4235653366281
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 539,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1420.7941742647351,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=538",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1420.7941742647351
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 916,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1397.0689576710797,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=915",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1397.0689576710797
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4.5-preview-2025-02-27",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 226,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1300.5645159074681,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=225",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1300.5645159074681
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 620,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1276.2539992799918,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=619",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1276.2539992799918
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 995,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1297.6612797837377,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=994",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1297.6612797837377
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4o-2024-05-13",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 502,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1453.5513805202027,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=501",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1453.5513805202027
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 69,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1437.1120977616538,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=68",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1437.1120977616538
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 851,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.6096326653767,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=850",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.6096326653767
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4.1-thinking",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 180,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1357.7670954967316,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=179",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1357.7670954967316
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 921,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1390.347905139712,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=920",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1390.347905139712
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2102,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1166.1721897810996,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=114",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1166.1721897810996
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mercury-2",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 190,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1342.0564568461446,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=189",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1342.0564568461446
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 563,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1380.412824085623,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=562",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1380.412824085623
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 938,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1373.8272084153182,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=937",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1373.8272084153182
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "minimax-m2",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 171,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1365.919647500117,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=170",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1365.919647500117
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 554,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1393.4723196308341,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=553",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1393.4723196308341
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 946,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1367.4297587014369,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=945",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1367.4297587014369
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o1-2024-12-17",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 157,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1376.178041830655,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=156",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1376.178041830655
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 882,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1425.743063431921,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=881",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1425.743063431921
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2227,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1371.0413869110375,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=239",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1371.0413869110375
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "solar-pro4",
      "numericRowCount": 6,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 6,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 970,
          "metricId": "Average",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.388",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=77;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.388
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 977,
          "metricId": "Average",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.472",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=85;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.472
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 3098,
          "metricId": "EM",
          "observedAt": "2023-09-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.457",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=222;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Average",
        "EM"
      ],
      "modelRef": "Baichuan2-13B-Chat",
      "numericRowCount": 6,
      "observedAtMax": "2023-09-06",
      "observedAtMin": "2023-09-06",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-trivia_qa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2305.10403",
          "line": 764,
          "metricId": "Challenge score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.692",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=39;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 69.19999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 732,
          "metricId": "Challenge score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.692",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=7;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 69.19999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1108,
          "metricId": "Score",
          "observedAt": "2023-05-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.909",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=172;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.909
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "PaLM 2-L",
      "numericRowCount": 6,
      "observedAtMax": "2023-05-17",
      "observedAtMin": "2023-05-17",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 820,
          "metricId": "Challenge score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.452",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=95;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2083,
          "metricId": "ECI Score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "102.37",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=454;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 102.37
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3071,
          "metricId": "EM",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.658",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=194;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 65.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Qwen2.5-Coder-1.5B",
      "numericRowCount": 6,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 827,
          "metricId": "Challenge score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.609",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=102;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2084,
          "metricId": "ECI Score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "112.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=456;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 112.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3078,
          "metricId": "EM",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.839",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=201;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 83.89999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Qwen2.5-Coder-7B",
      "numericRowCount": 6,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-mmlu_external",
        "epoch-the_agent_company_external",
        "hle",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2119,
          "metricId": "ECI Score",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "123.85",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=521;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 123.85
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3396,
          "metricId": "Mean score",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "6.05",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=32;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.05
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3640,
          "metricId": "EM",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.82",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=4;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 82.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "% Score",
        "Accuracy",
        "ECI Score",
        "EM",
        "Global average",
        "Mean score"
      ],
      "modelRef": "amazon.nova-pro-v1:0",
      "numericRowCount": 6,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-12-03",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-trivia_qa_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1970,
          "metricId": "ECI Score",
          "observedAt": "2023-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "119.51",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=275;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 119.51
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3533,
          "metricId": "mean_score",
          "observedAt": "2023-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1172583081570997",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=66;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.725830815709969
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www-cdn.anthropic.com/5c49cc247484cecf107c699baf29250302e5da70/claude-2-model-card.pdf",
          "line": 3653,
          "metricId": "EM",
          "observedAt": "2023-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.785",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=22;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 78.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "claude-2.0",
      "numericRowCount": 6,
      "observedAtMax": "2023-07-11",
      "observedAtMin": "2023-07-11",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1969,
          "metricId": "ECI Score",
          "observedAt": "2023-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.31",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=274;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.31
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2360,
          "metricId": "Overall score",
          "observedAt": "2023-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "54.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=77;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3654,
          "metricId": "EM",
          "observedAt": "2023-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.735",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=23;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 73.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "EM",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "claude-2.1",
      "numericRowCount": 6,
      "observedAtMax": "2023-11-21",
      "observedAtMin": "2023-11-21",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 77,
          "metricId": "Percent correct",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "64.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=68;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 64.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1964,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=265;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3504,
          "metricId": "mean_score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.9003021148036254",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=37;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 90.03021148036254
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_32K",
      "numericRowCount": 6,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-vpct_external",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1944,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=239;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3501,
          "metricId": "mean_score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.9116314199395771",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=34;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 91.16314199395771
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4196,
          "metricId": "mean_score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5777777777777777",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=189;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.77777777777777
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Correct",
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_64K",
      "numericRowCount": 6,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-trivia_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.anthropic.com/news/releasing-claude-instant-1-2",
          "line": 847,
          "metricId": "Challenge score",
          "observedAt": "2023-08-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.863",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=131;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 86.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2210,
          "metricId": "ECI Score",
          "observedAt": "2023-08-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "121.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=678;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 121.07
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.anthropic.com/news/releasing-claude-instant-1-2",
          "line": 3095,
          "metricId": "EM",
          "observedAt": "2023-08-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.867",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=219;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 86.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "claude-instant-1.2",
      "numericRowCount": 6,
      "observedAtMax": "2023-08-09",
      "observedAtMin": "2023-08-09",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1926,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=219;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.44
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2553,
          "metricId": "mean_score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.041666666666666664",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=60;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.166666666666666
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4164,
          "metricId": "mean_score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6888888888888889",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=157;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.88888888888889
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "claude-opus-4-1-20250805_27K",
      "numericRowCount": 6,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 86,
          "metricId": "Percent correct",
          "observedAt": "2025-10-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "74.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=77;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 74.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1300,
          "metricId": "mean_score",
          "observedAt": "2025-12-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.14",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=148;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.000000000000002
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1917,
          "metricId": "ECI Score",
          "observedAt": "2025-12-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "146.47",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=210;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 146.47
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "deepseek-reasoner",
      "numericRowCount": 6,
      "observedAtMax": "2025-12-01",
      "observedAtMin": "2025-10-03",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-mmlu_external",
        "epoch-simplebench_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 68,
          "metricId": "Percent correct",
          "observedAt": "2024-12-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "22.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=58;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 22.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2108,
          "metricId": "ECI Score",
          "observedAt": "2024-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=503;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.16
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3386,
          "metricId": "Mean score",
          "observedAt": "2024-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "7.15",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=22;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.15
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Global average",
        "Mean score",
        "Percent correct",
        "Score (AVG@5)"
      ],
      "modelRef": "gemini-2.0-flash-exp",
      "numericRowCount": 6,
      "observedAtMax": "2024-12-22",
      "observedAtMin": "2024-12-11",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 61,
          "metricId": "Percent correct",
          "observedAt": "2025-02-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "35.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=49;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1709,
          "metricId": "Accuracy",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0069",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=40;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.69
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1973,
          "metricId": "ECI Score",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.63",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=278;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.63
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Global average",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "gemini-2.0-pro-exp-02-05",
      "numericRowCount": 6,
      "observedAtMax": "2025-02-25",
      "observedAtMin": "2025-02-05",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-critpt_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 291,
          "metricId": "Score",
          "observedAt": "2026-02-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8458",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=15;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.8458
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 353,
          "metricId": "Score",
          "observedAt": "2026-02-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4514",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=77;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4514
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 506,
          "metricId": "Score",
          "observedAt": "2026-02-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.96",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=15;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.96
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "gemini-3-deep-think-preview",
      "numericRowCount": 6,
      "observedAtMax": "2026-02-12",
      "observedAtMin": "2026-02-12",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index",
        "epoch-mystery_game_puzzles",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1223,
          "metricId": "mean_score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=71;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1589,
          "metricId": "Average score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.479",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=22;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 47.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1800,
          "metricId": "ECI Score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "151.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=92;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 151.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Average score",
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemini-3-flash-preview_high",
      "numericRowCount": 6,
      "observedAtMax": "2025-12-17",
      "observedAtMin": "2025-12-17",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1967,
          "metricId": "ECI Score",
          "observedAt": "2024-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "122.42",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=272;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 122.42
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3554,
          "metricId": "mean_score",
          "observedAt": "2024-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2788897280966767",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=87;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.888972809667674
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3700,
          "metricId": "EM",
          "observedAt": "2024-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.757",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=73;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Global average",
        "mean_score"
      ],
      "modelRef": "gemma-2-27b-it",
      "numericRowCount": 6,
      "observedAtMax": "2024-06-24",
      "observedAtMin": "2024-06-24",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1968,
          "metricId": "ECI Score",
          "observedAt": "2024-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "119.52",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=273;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 119.52
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3553,
          "metricId": "mean_score",
          "observedAt": "2024-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2100641993957704",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=86;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.00641993957704
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3699,
          "metricId": "EM",
          "observedAt": "2024-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.721",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=72;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 72.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Global average",
        "mean_score"
      ],
      "modelRef": "gemma-2-9b-it",
      "numericRowCount": 6,
      "observedAtMax": "2024-06-24",
      "observedAtMin": "2024-06-24",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-simplebench_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 376,
          "metricId": "Score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1833",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=100;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.1833
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 599,
          "metricId": "Score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.7017",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=109;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.7017
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1673,
          "metricId": "Accuracy",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1875",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=4;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 18.75
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score",
        "Score (AVG@5)"
      ],
      "modelRef": "gpt-5-pro-2025-10-06_unknown",
      "numericRowCount": 6,
      "observedAtMax": "2025-10-07",
      "observedAtMin": "2025-10-07",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 854,
          "metricId": "Challenge score",
          "observedAt": "2022-04-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.411",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=140;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.099999999999994
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1133,
          "metricId": "Score",
          "observedAt": "2022-04-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.649",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=197;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.649
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 3200,
          "metricId": "Overall accuracy",
          "observedAt": "2022-04-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.705",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=65;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 70.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "gpt-neox-20b",
      "numericRowCount": 6,
      "observedAtMax": "2022-04-07",
      "observedAtMin": "2022-04-07",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-scicode_external",
        "epoch-weirdml_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1440,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0142857142857143",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=79;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.42857142857143
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1773,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.81",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=63;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.81
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4031,
          "metricId": "mean_score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5388888888888889",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=24;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.888888888888886
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "gpt-oss-20b_high",
      "numericRowCount": 6,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-apex_agents_external",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-simplebench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 274,
          "metricId": "Pass@1 score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.021",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=60;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 483,
          "metricId": "Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=207;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 716,
          "metricId": "Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.055",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=227;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.055
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Pass@1 score",
        "Score",
        "Score (AVG@5)"
      ],
      "modelRef": "grok-3",
      "numericRowCount": 6,
      "observedAtMax": "2025-04-09",
      "observedAtMin": "2025-04-09",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-forecastbench_external",
        "epoch-proofbench_external",
        "epoch-simplebench_external",
        "epoch-vending_bench_2_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 182,
          "metricId": "Performance",
          "observedAt": "2025-11-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "394.93",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=96;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 394.93
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2300,
          "metricId": "Overall score",
          "observedAt": "2025-11-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "61.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=17;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-proofbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4414,
          "metricId": "Accuracy",
          "observedAt": "2025-11-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "proofbench_external.csv:row=54;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Overall score",
        "Performance",
        "Score",
        "Score (AVG@5)"
      ],
      "modelRef": "grok-4-1-fast-reasoning",
      "numericRowCount": 6,
      "observedAtMax": "2026-01-05",
      "observedAtMin": "2025-11-19",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "epoch-webdev_arena_external",
        "gpqa-diamond",
        "simpleqa"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1308,
          "metricId": "mean_score",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=156;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 20.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1924,
          "metricId": "ECI Score",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "146.11",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=217;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 146.11
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4148,
          "metricId": "mean_score",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8305555555555556",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=141;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 83.05555555555556
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Arena Score",
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "kimi-k2-thinking-turbo",
      "numericRowCount": 6,
      "observedAtMax": "2026-01-05",
      "observedAtMin": "2025-11-06",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-apex_agents_external",
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-proofbench_external",
        "epoch-scicode_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 263,
          "metricId": "Pass@1 score",
          "observedAt": "2026-06-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.115",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=49;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1426,
          "metricId": "Accuracy",
          "observedAt": "2026-06-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0314285714285714",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=65;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.14285714285714
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2134,
          "metricId": "ECI Score",
          "observedAt": "2026-06-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "143.11",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=543;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 143.11
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Pass@1 score",
        "Score"
      ],
      "modelRef": "nemotron-3-ultra",
      "numericRowCount": 6,
      "observedAtMax": "2026-06-04",
      "observedAtMin": "2026-06-04",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 838,
          "metricId": "Challenge score",
          "observedAt": "2022-05-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.232",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=114;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 23.200000000000003
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 1101,
          "metricId": "Score",
          "observedAt": "2022-05-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.596",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=163;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.596
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 3230,
          "metricId": "Overall accuracy",
          "observedAt": "2022-05-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.415",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=102;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "opt-1.3b",
      "numericRowCount": 6,
      "observedAtMax": "2022-05-11",
      "observedAtMin": "2022-05-11",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 60,
          "metricId": "Percent correct",
          "observedAt": "2025-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "21.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=48;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1961,
          "metricId": "ECI Score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "133.39",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=260;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 133.39
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3498,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6718277945619335",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=31;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 67.18277945619336
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "ECI Score",
        "Percent correct",
        "mean_score"
      ],
      "modelRef": "qwen-max-2025-01-25",
      "numericRowCount": 6,
      "observedAtMax": "2025-01-28",
      "observedAtMin": "2025-01-25",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 859,
          "metricId": "Challenge score",
          "observedAt": "2023-04-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.27",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=145;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 1138,
          "metricId": "Score",
          "observedAt": "2023-04-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.59",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=202;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.59
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.03450",
          "line": 3252,
          "metricId": "Overall accuracy",
          "observedAt": "2023-04-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.407",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=124;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.699999999999996
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "stablelm-tuned-alpha-7b",
      "numericRowCount": 6,
      "observedAtMax": "2023-04-19",
      "observedAtMin": "2023-04-19",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 6,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2303.08774",
          "line": 841,
          "metricId": "Challenge score",
          "observedAt": "2022-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.852",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=121;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 85.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1033,
          "metricId": "Score",
          "observedAt": "2022-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.877",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=29;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.877
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2946,
          "metricId": "EM",
          "observedAt": "2022-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.415",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=9;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "text-davinci-002",
      "numericRowCount": 6,
      "observedAtMax": "2022-03-15",
      "observedAtMin": "2022-03-15",
      "rowCount": 6,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 342,
          "metricId": "resolved",
          "observedAt": "2024-04-02",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Lite",
            "scaffold": "SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 11.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[77]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 11.67
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 343,
          "metricId": "resolved",
          "observedAt": "2024-04-02",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Lite",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 4.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[78]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 4.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 78,
          "metricId": "resolved",
          "observedAt": "2024-04-02",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Test",
            "scaffold": "SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 10.51,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[17]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 10.51
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 6
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 3 Opus",
      "numericRowCount": 6,
      "observedAtMax": "2024-04-02",
      "observedAtMin": "2024-04-02",
      "rowCount": 6,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 6
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 299,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1194.8091498295357,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=298",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1194.8091498295357
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 689,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1153.7295381330277,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=688",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1153.7295381330277
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1139,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 950.369134936258,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=138",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 950.369134936258
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-haiku-20240307",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 267,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1262.1333333140765,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=266",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1262.1333333140765
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 638,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1246.773068977299,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=637",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1246.773068977299
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1124,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1023.141220485738,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=123",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1023.141220485738
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-opus-20240229",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 289,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1218.1700745633068,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=288",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1218.1700745633068
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 679,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1187.6305164071323,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=678",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1187.6305164071323
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1134,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 984.1332078913672,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=133",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 984.1332078913672
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-3-sonnet-20240229",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 275,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1239.4318473523422,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=274",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1239.4318473523422
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 652,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1233.2314537743384,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=651",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1233.2314537743384
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1127,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1008.1734659872603,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=126",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1008.1734659872603
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-1.5-flash-001",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 240,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1286.774645635874,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=239",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1286.774645635874
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 616,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1293.8435589375351,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=615",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1293.8435589375351
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1103,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1140.858488233915,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=102",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1140.858488233915
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-1.5-flash-002",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 285,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1225.9813763400402,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=284",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1225.9813763400402
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 653,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1229.8890146570743,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=652",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1229.8890146570743
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1123,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1044.3197343349811,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=122",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1044.3197343349811
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-1.5-flash-8b-001",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 256,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1273.6244525029747,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=255",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1273.6244525029747
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 622,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1273.7686022750856,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=621",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1273.7686022750856
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1115,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1088.5156027964017,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=114",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1088.5156027964017
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-1.5-pro-001",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 257,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1271.8197311485192,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=256",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1271.8197311485192
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 643,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1241.2370706557867,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=642",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1241.2370706557867
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1113,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1090.0807934474026,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=112",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1090.0807934474026
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4-turbo-2024-04-09",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 242,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1284.8081205181213,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=241",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1284.8081205181213
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 627,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1270.10550676687,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=626",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1270.10550676687
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 991,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1306.220123689388,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=990",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1306.220123689388
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4.1-nano-2025-04-14",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 245,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1282.6687712680166,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=244",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1282.6687712680166
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 633,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1253.152500400838,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=632",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1253.152500400838
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1118,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1064.7664895720784,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=117",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1064.7664895720784
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4o-2024-08-06",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 241,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1286.6282361072251,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=240",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1286.6282361072251
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 629,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1263.1366246553691,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=628",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1263.1366246553691
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1117,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1065.8332358257553,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=116",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1065.8332358257553
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4o-mini-2024-07-18",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 231,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1295.4800148046925,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=230",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1295.4800148046925
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1116,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1077.244855183787,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=115",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1077.244855183787
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1512,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "diagram",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1065.5321906384077,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=511",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1065.5321906384077
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "molmo-2-8b",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1313,
          "metricId": "arena_score_bt",
          "observedAt": "2026-01-09",
          "protocol": {
            "arena_config": "vision",
            "category": "creative_writing",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1183.446886353961,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=312",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1183.446886353961
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1452,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "diagram",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1266.9377133004402,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=451",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1266.9377133004402
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1579,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1225.1673790768318,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=578",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1225.1673790768318
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen-vl-max-2025-08-13",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-01-09",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2126,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-25",
          "protocol": {
            "arena_config": "webdev",
            "category": "image_to_webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1507.9023862615309,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=138",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1507.9023862615309
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2175,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1521.44604631191,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=187",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1521.44604631191
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2013,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1521.44604631191,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=25",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1521.44604631191
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "seed-2.1-pro-preview",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-25",
      "observedAtMin": "2026-08-21",
      "rowCount": 5,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 925,
          "metricId": "Average",
          "observedAt": "2023-06-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3248",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=30;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3248
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2219,
          "metricId": "ECI Score",
          "observedAt": "2023-06-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "89.59",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=746;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 89.59
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 2988,
          "metricId": "EM",
          "observedAt": "2023-06-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.092",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=90;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "Baichuan-7B",
      "numericRowCount": 5,
      "observedAtMax": "2023-06-01",
      "observedAtMin": "2023-06-01",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1985,
          "metricId": "ECI Score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "113.57",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=290;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 113.57
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2364,
          "metricId": "Overall score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "51.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=81;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 51.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3549,
          "metricId": "mean_score",
          "observedAt": "2023-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.032854984894259816",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=82;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.2854984894259815
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "Llama-2-70b-chat-hf",
      "numericRowCount": 5,
      "observedAtMax": "2023-07-18",
      "observedAtMin": "2023-07-18",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1981,
          "metricId": "ECI Score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "122.56",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=286;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 122.56
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3548,
          "metricId": "mean_score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.225547583081571",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=81;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 22.5547583081571
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3749,
          "metricId": "EM",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.793",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=129;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "Meta-Llama-3-70B-Instruct",
      "numericRowCount": 5,
      "observedAtMax": "2024-04-18",
      "observedAtMin": "2024-04-18",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 818,
          "metricId": "Challenge score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.344",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=93;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 34.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3069,
          "metricId": "EM",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.345",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=192;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 34.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2409.12186",
          "line": 3244,
          "metricId": "Overall accuracy",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.484",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=116;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 48.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Qwen2.5-Coder-0.5B",
      "numericRowCount": 5,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 829,
          "metricId": "Challenge score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.66",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=104;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 66.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3080,
          "metricId": "EM",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.887",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=203;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 88.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2409.12186",
          "line": 3248,
          "metricId": "Overall accuracy",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.802",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=120;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 80.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Qwen2.5-Coder-14B",
      "numericRowCount": 5,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external",
        "epoch-spatialviz_bench_external",
        "osworld"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2163,
          "metricId": "ECI Score",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "129.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=613;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 129.07
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-geobench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://geobench.org/",
          "line": 2664,
          "metricId": "ACW Avg Score",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "3448",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "geobench_external.csv:row=22;column=ACW Avg Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 3448.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-spatialviz_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4770,
          "metricId": "Overall score",
          "observedAt": "2024-09-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3331",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "spatialviz_bench_external.csv:row=2;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 33.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ACW Avg Score",
        "ECI Score",
        "Overall score",
        "Score"
      ],
      "modelRef": "Qwen2.5-VL-72B-Instruct",
      "numericRowCount": 5,
      "observedAtMax": "2024-09-19",
      "observedAtMin": "2024-09-19",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 81,
          "metricId": "Percent correct",
          "observedAt": "2025-05-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=72;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 455,
          "metricId": "Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0125",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=179;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0125
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 708,
          "metricId": "Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.11",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=219;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.11
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Score"
      ],
      "modelRef": "Qwen3-235B-A22B-Instruct-2507",
      "numericRowCount": 5,
      "observedAtMax": "2025-07-25",
      "observedAtMin": "2025-05-09",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 910,
          "metricId": "Average",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.543",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=15;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.543
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2220,
          "metricId": "ECI Score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "117.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=748;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 117.04
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 3015,
          "metricId": "EM",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.672",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=121;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 67.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Average",
        "ECI Score",
        "EM"
      ],
      "modelRef": "Yi-34B",
      "numericRowCount": 5,
      "observedAtMax": "2023-11-02",
      "observedAtMin": "2023-11-02",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 819,
          "metricId": "Challenge score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.254",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=94;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2222,
          "metricId": "ECI Score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "62.85",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=754;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 62.85
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3070,
          "metricId": "EM",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.044",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=193;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.3999999999999995
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "deepseek-coder-1.3b-base",
      "numericRowCount": 5,
      "observedAtMax": "2023-11-02",
      "observedAtMin": "2023-11-02",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 830,
          "metricId": "Challenge score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.422",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=105;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.199999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2227,
          "metricId": "ECI Score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "95.57",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=761;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 95.57
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3081,
          "metricId": "EM",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.354",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=204;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "deepseek-coder-33b-base",
      "numericRowCount": 5,
      "observedAtMax": "2023-11-02",
      "observedAtMin": "2023-11-02",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 824,
          "metricId": "Challenge score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.364",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=99;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 36.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2225,
          "metricId": "ECI Score",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "88.75",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=757;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 88.75
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3075,
          "metricId": "EM",
          "observedAt": "2023-11-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.213",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=198;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "deepseek-coder-6.7b-base",
      "numericRowCount": 5,
      "observedAtMax": "2023-11-02",
      "observedAtMin": "2023-11-02",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1989,
          "metricId": "ECI Score",
          "observedAt": "2023-12-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "116.93",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=295;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 116.93
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3561,
          "metricId": "mean_score",
          "observedAt": "2023-12-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.11244335347432025",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=94;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.244335347432024
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3689,
          "metricId": "EM",
          "observedAt": "2023-12-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=61;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 70.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "gemini-1.0-pro-001",
      "numericRowCount": 5,
      "observedAtMax": "2023-12-13",
      "observedAtMin": "2023-12-13",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1221,
          "metricId": "mean_score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.25",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=69;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1605,
          "metricId": "Average score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.373",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=38;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 37.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1798,
          "metricId": "ECI Score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "145.15",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=90;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 145.15
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Average score",
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemini-3.1-flash-lite_low",
      "numericRowCount": 5,
      "observedAtMax": "2026-03-03",
      "observedAtMin": "2026-03-03",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 134,
          "metricId": "Performance",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "927.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=48;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 927.17
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1473,
          "metricId": "Accuracy",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=112;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4464,
          "metricId": "Score",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4004629629629631",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=32;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4004629629629631
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Performance",
        "Score"
      ],
      "modelRef": "gemma-4-26b-a4b",
      "numericRowCount": 5,
      "observedAtMax": "2026-04-02",
      "observedAtMin": "2026-04-02",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "epoch-simplebench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 598,
          "metricId": "Score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.702",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=108;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.702
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1878,
          "metricId": "ECI Score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=170;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.29
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2546,
          "metricId": "mean_score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.146",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=53;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "gpt-5-pro-2025-10-06_high",
      "numericRowCount": 5,
      "observedAtMax": "2025-10-07",
      "observedAtMin": "2025-10-07",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-proofbench_external",
        "epoch-scicode_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1410,
          "metricId": "Accuracy",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.08285714285714291",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=49;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.28571428571429
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2132,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=540;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.17
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-proofbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4412,
          "metricId": "Accuracy",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.06",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "proofbench_external.csv:row=52;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "Score"
      ],
      "modelRef": "inkling-small",
      "numericRowCount": 5,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 972,
          "metricId": "Average",
          "observedAt": "2023-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.424",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=79;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.424
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 979,
          "metricId": "Average",
          "observedAt": "2023-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.367",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=87;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.367
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 3100,
          "metricId": "EM",
          "observedAt": "2023-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.157",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=224;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Average",
        "EM"
      ],
      "modelRef": "internlm-chat-20b",
      "numericRowCount": 5,
      "observedAtMax": "2023-09-17",
      "observedAtMin": "2023-09-17",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 485,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=209;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 719,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.05",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=230;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.05
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1811,
          "metricId": "ECI Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "134.18",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=103;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 134.18
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "magistral-small-2506",
      "numericRowCount": 5,
      "observedAtMax": "2025-06-10",
      "observedAtMin": "2025-06-10",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 148,
          "metricId": "Performance",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "785.58",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=62;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 785.58
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1452,
          "metricId": "Accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00848979591836735",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=91;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.8489795918367351
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4523,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.38657407407407407",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=91;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.38657407407407407
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Performance",
        "Score"
      ],
      "modelRef": "mercury-2",
      "numericRowCount": 5,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1987,
          "metricId": "ECI Score",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "120.97",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=293;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 120.97
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3569,
          "metricId": "mean_score",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.24461858006042297",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=102;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 24.4618580060423
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3785,
          "metricId": "EM",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.688",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=165;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "mean_score"
      ],
      "modelRef": "mistral-large-2402",
      "numericRowCount": 5,
      "observedAtMax": "2024-02-26",
      "observedAtMin": "2024-02-26",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-deepswe_external",
        "epoch-epoch_capabilities_index",
        "epoch-gdp_pdf_external",
        "epoch-webdev_arena_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepswe_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1670,
          "metricId": "Pass@1",
          "observedAt": "2026-08-05",
          "protocol": {
            "harness": "mini-swe-agent",
            "subject_type": "system"
          },
          "rawValue": "0.5486725663716814",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepswe_external.csv:row=62;column=Pass@1",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.86725663716814
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2236,
          "metricId": "ECI Score",
          "observedAt": "2026-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "155.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=781;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 155.16
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gdp_pdf_external",
          "benchmarkName": null,
          "evidenceUrl": "https://surgehq.ai/benchmarks/gdp-pdf",
          "line": 2631,
          "metricId": "GDP.pdf score",
          "observedAt": "2026-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gdp_pdf_external.csv:row=29;column=GDP.pdf score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.16
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "ECI Score",
        "GDP.pdf score",
        "Pass@1"
      ],
      "modelRef": "muse-spark-1.2_xhigh",
      "numericRowCount": 5,
      "observedAtMax": "2026-08-05",
      "observedAtMin": "2026-08-05",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_agi_external",
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 672,
          "metricId": "Score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.272",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=183;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.272
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1245,
          "metricId": "mean_score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.07",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=93;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.000000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1842,
          "metricId": "ECI Score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.67",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=134;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "o1-2024-12-17_low",
      "numericRowCount": 5,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-12-17",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "frontiermath",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1975,
          "metricId": "ECI Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=280;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.64
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3514,
          "metricId": "mean_score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8918051359516617",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=47;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 89.18051359516616
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4206,
          "metricId": "mean_score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.46944444444444444",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=199;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.94444444444444
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "o1-mini-2024-09-12_high",
      "numericRowCount": 5,
      "observedAtMax": "2024-09-12",
      "observedAtMin": "2024-09-12",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 28,
          "metricId": "Percent correct",
          "observedAt": "2025-06-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "84.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=15;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 84.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 408,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0486",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=132;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0486
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 612,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5933",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=122;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.5933
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct",
        "Score"
      ],
      "modelRef": "o3-pro-2025-06-10_high",
      "numericRowCount": 5,
      "observedAtMax": "2025-06-28",
      "observedAtMin": "2025-06-10",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "epoch-math_level_5",
        "epoch-simplebench_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1946,
          "metricId": "ECI Score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "139.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=241;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 139.64
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3367,
          "metricId": "Mean score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "8.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=3;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3479,
          "metricId": "mean_score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6885699899295065",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=12;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.85699899295065
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "ECI Score",
        "Mean score",
        "Score (AVG@5)",
        "mean_score"
      ],
      "modelRef": "qwen3-235b-a22b",
      "numericRowCount": 5,
      "observedAtMax": "2025-04-29",
      "observedAtMin": "2025-04-29",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 828,
          "metricId": "Challenge score",
          "observedAt": "2024-02-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.472",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=103;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 47.199999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2226,
          "metricId": "ECI Score",
          "observedAt": "2024-02-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "104.43",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=760;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 104.43
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3079,
          "metricId": "EM",
          "observedAt": "2024-02-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.577",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=202;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.699999999999996
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "starcoder2-15b",
      "numericRowCount": 5,
      "observedAtMax": "2024-02-20",
      "observedAtMin": "2024-02-20",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 821,
          "metricId": "Challenge score",
          "observedAt": "2024-02-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.342",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=96;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 34.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2223,
          "metricId": "ECI Score",
          "observedAt": "2024-02-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "87.88",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=755;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 87.88
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3072,
          "metricId": "EM",
          "observedAt": "2024-02-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.216",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=195;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "starcoder2-3b",
      "numericRowCount": 5,
      "observedAtMax": "2024-02-22",
      "observedAtMin": "2024-02-22",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 5,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 823,
          "metricId": "Challenge score",
          "observedAt": "2024-02-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.387",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=98;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 38.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2224,
          "metricId": "ECI Score",
          "observedAt": "2024-02-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "92.75",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=756;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 92.75
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3074,
          "metricId": "EM",
          "observedAt": "2024-02-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.327",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=197;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "ECI Score",
        "EM"
      ],
      "modelRef": "starcoder2-7b",
      "numericRowCount": 5,
      "observedAtMax": "2024-02-20",
      "observedAtMin": "2024-02-20",
      "rowCount": 5,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 282,
          "metricId": "resolved",
          "observedAt": "2025-09-01",
          "protocol": {
            "checked": null,
            "harness": "EntroPO + R2E",
            "leaderboard_variant": "Lite",
            "scaffold": "EntroPO + R2E",
            "subject_type": "model"
          },
          "rawValue": 45.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[17]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 45.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 272,
          "metricId": "resolved",
          "observedAt": "2025-09-01",
          "protocol": {
            "checked": null,
            "harness": "EntroPO + R2E",
            "leaderboard_variant": "Lite",
            "scaffold": "EntroPO + R2E",
            "subject_type": "model"
          },
          "rawValue": 49.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[7]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 49.67
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 187,
          "metricId": "resolved",
          "observedAt": "2025-09-01",
          "protocol": {
            "checked": null,
            "harness": "EntroPO + R2E",
            "leaderboard_variant": "Verified",
            "scaffold": "EntroPO + R2E",
            "subject_type": "model"
          },
          "rawValue": 52.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[102]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 52.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 5
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Qwen3-Coder-30B-A3B-Instruct",
      "numericRowCount": 5,
      "observedAtMax": "2025-09-01",
      "observedAtMin": "2025-08-05",
      "rowCount": 5,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 5
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 281,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1228.5939282784916,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=280",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1228.5939282784916
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 656,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1224.5877805700463,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=655",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1224.5877805700463
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1132,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 990.6582197440717,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=131",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 990.6582197440717
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-lite-v1.0",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-text",
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 269,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1258.7787203480088,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=268",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1258.7787203480088
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 640,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1242.9498057221656,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=639",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1242.9498057221656
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1135,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 980.6633316517574,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=134",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 980.6633316517574
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-pro-v1.0",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2100,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1193.5122127627499,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=112",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1193.5122127627499
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2262,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1193.5122127627499,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=274",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1193.5122127627499
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2380,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-html",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1216.2233841796915,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=392",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1216.2233841796915
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "devstral-2",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-search",
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2492,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1226.9585452946385,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=5",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1226.9585452946385
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 24,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1468.0783615539829,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=23",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1468.0783615539829
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 433,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1503.7210502502985,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=432",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1503.7210502502985
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ernie-5.1",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "arena-search",
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2504,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1189.24344568097,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=17",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1189.24344568097
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 457,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1476.7340041003833,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=456",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1476.7340041003833
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 53,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.4283470613332,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=52",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.4283470613332
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4.20-beta1",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2237,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1347.3095672093864,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=249",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1347.3095672093864
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2361,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-html",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1323.9567930741878,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=373",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1323.9567930741878
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2468,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-react",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1338.4811199927535,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=480",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1338.4811199927535
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "laguna-m.1",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2246,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1301.9706358413464,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=258",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1301.9706358413464
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2371,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-html",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1279.5120495419758,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=383",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1279.5120495419758
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2475,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-react",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1294.8834584918536,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=487",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1294.8834584918536
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "laguna-xs.2",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 4,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 826,
          "metricId": "Challenge score",
          "observedAt": "2024-04-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.357",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=101;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.699999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3077,
          "metricId": "EM",
          "observedAt": "2024-04-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.377",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=200;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 37.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3665,
          "metricId": "EM",
          "observedAt": "2024-04-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.405",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=34;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM"
      ],
      "modelRef": "CodeQwen1.5-7B",
      "numericRowCount": 4,
      "observedAtMax": "2024-04-15",
      "observedAtMin": "2024-04-15",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 825,
          "metricId": "Challenge score",
          "observedAt": "2024-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.573",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=100;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3076,
          "metricId": "EM",
          "observedAt": "2024-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.671",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=199;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 67.10000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3676,
          "metricId": "EM",
          "observedAt": "2024-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.605",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=47;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 60.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM"
      ],
      "modelRef": "DeepSeek-Coder-V2-Lite-Base",
      "numericRowCount": 4,
      "observedAtMax": "2024-06-13",
      "observedAtMin": "2024-06-13",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3506,
          "metricId": "mean_score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8989803625377644",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=39;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 89.89803625377644
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4201,
          "metricId": "mean_score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5138888888888888",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=194;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 51.388888888888886
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2865,
          "metricId": "mean_score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5574494949494949",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=191;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.744949494949495
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Global average",
        "mean_score"
      ],
      "modelRef": "DeepSeek-R1-Distill-Llama-70B",
      "numericRowCount": 4,
      "observedAtMax": "2025-01-20",
      "observedAtMin": "2025-01-20",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-lambada_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 1119,
          "metricId": "Score",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.897",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=183;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.897
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lambada_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 3362,
          "metricId": "Score",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.785",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lambada_external.csv:row=78;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.785
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://inflection.ai/blog/inflection-1",
          "line": 3724,
          "metricId": "EM",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.727",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=100;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 72.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "EM",
        "Score"
      ],
      "modelRef": "Inflection-1",
      "numericRowCount": 4,
      "observedAtMax": "2023-06-22",
      "observedAtMin": "2023-06-22",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.08295",
          "line": 1028,
          "metricId": "Score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.832",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=23;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.832
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.08295",
          "line": 2971,
          "metricId": "EM",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.354",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=73;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2403.08295",
          "line": 3780,
          "metricId": "EM",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.625",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=160;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 62.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "EM",
        "Score"
      ],
      "modelRef": "Mistral-7B-Instruct-v0.2",
      "numericRowCount": 4,
      "observedAtMax": "2023-12-11",
      "observedAtMin": "2023-12-11",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1063,
          "metricId": "Score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.825",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=123;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.825
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2140,
          "metricId": "ECI Score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.31",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=555;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.31
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 2998,
          "metricId": "EM",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.842",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=102;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 84.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "Mistral-Nemo-Base-2407",
      "numericRowCount": 4,
      "observedAtMax": "2024-07-18",
      "observedAtMin": "2024-07-18",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2006,
          "metricId": "ECI Score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=317;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.04
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2355,
          "metricId": "Overall score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "56.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=72;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3557,
          "metricId": "mean_score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.09290030211480363",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=90;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.290030211480364
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Overall score",
        "mean_score"
      ],
      "modelRef": "Mixtral-8x7B-Instruct-v0.1",
      "numericRowCount": 4,
      "observedAtMax": "2023-12-11",
      "observedAtMin": "2023-12-11",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-gsm8k_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 55,
          "metricId": "Percent correct",
          "observedAt": "2024-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "8.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=43;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 70,
          "metricId": "Percent correct",
          "observedAt": "2024-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "16.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=60;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 16.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3068,
          "metricId": "EM",
          "observedAt": "2024-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.93",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=191;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 93.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "EM",
        "Global average",
        "Percent correct"
      ],
      "modelRef": "Qwen2.5-Coder-32B-Instruct",
      "numericRowCount": 4,
      "observedAtMax": "2024-11-21",
      "observedAtMin": "2024-11-21",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 822,
          "metricId": "Challenge score",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.529",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=97;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.900000000000006
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3073,
          "metricId": "EM",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.757",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=196;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2409.12186",
          "line": 3246,
          "metricId": "Overall accuracy",
          "observedAt": "2024-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.709",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=118;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 70.89999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Qwen2.5-Coder-3B",
      "numericRowCount": 4,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/",
          "line": 17,
          "metricId": "Percent correct",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=3;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2099,
          "metricId": "ECI Score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "139.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=488;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 139.64
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2320,
          "metricId": "Overall score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=37;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score",
        "Percent correct"
      ],
      "modelRef": "Qwen3-235B-A22B",
      "numericRowCount": 4,
      "observedAtMax": "2025-04-29",
      "observedAtMin": "2025-04-29",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-common_sense_qa_2_external",
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 1143,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.912",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=207;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.912
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1044,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.761",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=54;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.761
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-common_sense_qa_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.05320",
          "line": 1359,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.678",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "common_sense_qa_2_external.csv:row=5;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.678
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "T5-11B",
      "numericRowCount": 4,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-common_sense_qa_2_external",
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 1141,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.854",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=205;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.854
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-common_sense_qa_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2201.05320",
          "line": 1358,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.546",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "common_sense_qa_2_external.csv:row=3;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.546
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 4785,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.823",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=10;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.823
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "T5-Large",
      "numericRowCount": 4,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-hella_swag_external",
        "epoch-mmlu_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 728,
          "metricId": "Challenge score",
          "observedAt": "2024-03-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.556",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=3;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.60000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 3261,
          "metricId": "Overall accuracy",
          "observedAt": "2024-03-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.764",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=136;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 76.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2403.04652",
          "line": 3850,
          "metricId": "EM",
          "observedAt": "2024-03-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.684",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=246;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM",
        "Overall accuracy"
      ],
      "modelRef": "Yi-9B",
      "numericRowCount": 4,
      "observedAtMax": "2024-03-01",
      "observedAtMin": "2024-03-01",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-mmlu_external",
        "hle",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 192,
          "metricId": "Performance",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "236.25",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=106;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 236.25
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3638,
          "metricId": "EM",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=2;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3311,
          "metricId": "Accuracy",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0364",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hle_external.csv:row=51;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.64
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "EM",
        "Global average",
        "Performance"
      ],
      "modelRef": "amazon.nova-lite-v1:0",
      "numericRowCount": 4,
      "observedAtMax": "2024-12-03",
      "observedAtMin": "2024-12-03",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 438,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0198",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=162;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0198
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 654,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3333",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=164;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3333
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2187,
          "metricId": "ECI Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=639;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20_16K",
      "numericRowCount": 4,
      "observedAtMax": "2025-05-20",
      "observedAtMin": "2025-05-20",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 32,
          "metricId": "Percent correct",
          "observedAt": "2025-05-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "55.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=19;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 429,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.025400000000000002",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=153;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.025400000000000002
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 660,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3233",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=171;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3233
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Percent correct",
        "Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20_23K",
      "numericRowCount": 4,
      "observedAtMax": "2025-05-25",
      "observedAtMin": "2025-05-20",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 487,
          "metricId": "Score",
          "observedAt": "2025-06-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=211;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 662,
          "metricId": "Score",
          "observedAt": "2025-06-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.313",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=173;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.313
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 695,
          "metricId": "Score",
          "observedAt": "2025-06-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=206;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.16
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "gemini-2.5-pro-preview-06-05_1K",
      "numericRowCount": 4,
      "observedAtMax": "2025-06-05",
      "observedAtMin": "2025-06-05",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 415,
          "metricId": "Score",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0403",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=139;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0403
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 641,
          "metricId": "Score",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.41000000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=151;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.41000000000000003
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2183,
          "metricId": "ECI Score",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "145.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=634;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 145.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score"
      ],
      "modelRef": "gemini-2.5-pro_16K",
      "numericRowCount": 4,
      "observedAtMax": "2025-06-17",
      "observedAtMin": "2025-06-17",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1222,
          "metricId": "mean_score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=70;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 20.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1799,
          "metricId": "ECI Score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "145.15",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=91;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 145.15
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4069,
          "metricId": "mean_score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=62;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 80.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemini-3.1-flash-lite_high",
      "numericRowCount": 4,
      "observedAtMax": "2026-03-03",
      "observedAtMin": "2026-03-03",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1220,
          "metricId": "mean_score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.24",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=68;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 24.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1797,
          "metricId": "ECI Score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "145.15",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=89;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 145.15
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4067,
          "metricId": "mean_score",
          "observedAt": "2026-03-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.37777777777777777",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=60;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 37.77777777777778
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemini-3.1-flash-lite_minimal",
      "numericRowCount": 4,
      "observedAtMax": "2026-03-03",
      "observedAtMin": "2026-03-03",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-simplebench_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 57,
          "metricId": "Percent correct",
          "observedAt": "2024-12-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "38.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=45;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 38.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2107,
          "metricId": "ECI Score",
          "observedAt": "2024-12-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=500;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.16
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-simplebench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4667,
          "metricId": "Score (AVG@5)",
          "observedAt": "2024-12-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.311",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "simplebench_external.csv:row=81;column=Score (AVG@5)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.311
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Global average",
        "Percent correct",
        "Score (AVG@5)"
      ],
      "modelRef": "gemini-exp-1206",
      "numericRowCount": 4,
      "observedAtMax": "2024-12-22",
      "observedAtMin": "2024-12-06",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1065,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.857",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=125;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.857
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2141,
          "metricId": "ECI Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "119.52",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=556;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 119.52
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 3000,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.849",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=104;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 84.89999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "gemma-2-9b",
      "numericRowCount": 4,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-chess_puzzles",
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-chess_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1211,
          "metricId": "mean_score",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.05",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "chess_puzzles.csv:row=59;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 5.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1788,
          "metricId": "ECI Score",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=80;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4057,
          "metricId": "mean_score",
          "observedAt": "2026-04-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.7333333333333333",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=50;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 73.33333333333333
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemma-4-31b-it_minimal",
      "numericRowCount": 4,
      "observedAtMax": "2026-04-02",
      "observedAtMin": "2026-04-02",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2001,
          "metricId": "ECI Score",
          "observedAt": "2024-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "125.98",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=307;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 125.98
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3552,
          "metricId": "mean_score",
          "observedAt": "2024-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3541351963746224",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=85;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.41351963746224
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "line": 3616,
          "metricId": "average_score",
          "observedAt": "2024-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.351726",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=41;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.351726
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4-0125-preview",
      "numericRowCount": 4,
      "observedAtMax": "2024-01-25",
      "observedAtMin": "2024-01-25",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "epoch-metr_time_horizons_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2003,
          "metricId": "ECI Score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "125.98",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=309;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 125.98
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3530,
          "metricId": "mean_score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.40020770392749244",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=63;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.02077039274924
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "line": 3619,
          "metricId": "average_score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.28905",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=44;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.28905
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "average_score",
        "mean_score"
      ],
      "modelRef": "gpt-4-1106-preview",
      "numericRowCount": 4,
      "observedAtMax": "2023-11-06",
      "observedAtMin": "2023-11-06",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-hella_swag_external",
        "epoch-open_book_qa_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 1102,
          "metricId": "Score",
          "observedAt": "2023-03-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.618",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=164;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.618
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 3199,
          "metricId": "Overall accuracy",
          "observedAt": "2023-03-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.427",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=63;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.699999999999996
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-open_book_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.05463",
          "line": 3922,
          "metricId": "Accuracy",
          "observedAt": "2023-03-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.232",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "open_book_qa_external.csv:row=16;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 23.200000000000003
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "gpt-neo-2.7B",
      "numericRowCount": 4,
      "observedAtMax": "2023-03-30",
      "observedAtMin": "2023-03-30",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-terminalbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 173,
          "metricId": "Performance",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "566.05",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=87;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 566.05
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2213,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.81",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=687;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.81
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 5046,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.034",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=202;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.4000000000000004
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy mean",
        "ECI Score",
        "Performance"
      ],
      "modelRef": "gpt-oss-20b",
      "numericRowCount": 4,
      "observedAtMax": "2025-11-03",
      "observedAtMin": "2025-08-05",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 486,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=210;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 492,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=216;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 711,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.061200000000000004",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=222;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.061200000000000004
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "magistral-medium-2506",
      "numericRowCount": 4,
      "observedAtMax": "2025-06-10",
      "observedAtMin": "2025-06-10",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 191,
          "metricId": "Performance",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "264.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=105;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 264.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1487,
          "metricId": "Accuracy",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=126;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4525,
          "metricId": "Score",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.362268518518519",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=93;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.362268518518519
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Performance",
        "Score"
      ],
      "modelRef": "mistral-large-2512",
      "numericRowCount": 4,
      "observedAtMax": "2025-12-02",
      "observedAtMin": "2025-12-02",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 195,
          "metricId": "Performance",
          "observedAt": "2025-08-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "210.18",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=109;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 210.18
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1480,
          "metricId": "Accuracy",
          "observedAt": "2025-08-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=119;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4530,
          "metricId": "Score",
          "observedAt": "2025-08-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3379629629629631",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=98;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3379629629629631
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Performance",
        "Score"
      ],
      "modelRef": "mistral-medium-2508",
      "numericRowCount": 4,
      "observedAtMax": "2025-08-12",
      "observedAtMin": "2025-08-12",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3524,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4481684290030212",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=57;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.81684290030212
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4194,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.05277777777777778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=187;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 5.277777777777778
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2881,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4529671717171717",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=207;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.29671717171717
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Global average",
        "mean_score"
      ],
      "modelRef": "mistral-small-2501",
      "numericRowCount": 4,
      "observedAtMax": "2025-01-25",
      "observedAtMin": "2025-01-25",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-cl_bench_external",
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "frontiermath"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-cl_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1331,
          "metricId": "Overall",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.176",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "cl_bench_external.csv:row=16;column=Overall",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.176
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1928,
          "metricId": "ECI Score",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "146.11",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=221;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 146.11
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2533,
          "metricId": "mean_score",
          "observedAt": "2025-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=40;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Overall",
        "mean_score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Thinking",
      "numericRowCount": 4,
      "observedAtMax": "2025-11-06",
      "observedAtMin": "2025-11-06",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-gdp_pdf_external",
        "epoch-proofbench_external",
        "epoch-simplebench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2130,
          "metricId": "ECI Score",
          "observedAt": "2026-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "155.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=534;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 155.16
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gdp_pdf_external",
          "benchmarkName": null,
          "evidenceUrl": "https://surgehq.ai/benchmarks/gdp-pdf",
          "line": 2632,
          "metricId": "GDP.pdf score",
          "observedAt": "2026-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gdp_pdf_external.csv:row=30;column=GDP.pdf score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.12
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-proofbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4380,
          "metricId": "Accuracy",
          "observedAt": "2026-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.43",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "proofbench_external.csv:row=20;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "GDP.pdf score",
        "Score (AVG@5)"
      ],
      "modelRef": "muse-spark-1.2",
      "numericRowCount": 4,
      "observedAtMax": "2026-08-05",
      "observedAtMin": "2026-08-05",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 194,
          "metricId": "Performance",
          "observedAt": "2026-03-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "213.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=108;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 213.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1428,
          "metricId": "Accuracy",
          "observedAt": "2026-03-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0314285714285714",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=67;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.14285714285714
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4527,
          "metricId": "Score",
          "observedAt": "2026-03-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.35995370370370405",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=95;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.35995370370370405
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Performance",
        "Score"
      ],
      "modelRef": "nemotron-3-super",
      "numericRowCount": 4,
      "observedAtMax": "2026-03-11",
      "observedAtMin": "2026-03-11",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index",
        "epoch-spatialviz_bench_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1690,
          "metricId": "Accuracy",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0565",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=21;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 5.65
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2234,
          "metricId": "ECI Score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.67",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=777;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.67
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-spatialviz_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4773,
          "metricId": "Overall score",
          "observedAt": "2024-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4136",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "spatialviz_bench_external.csv:row=5;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.36
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score"
      ],
      "modelRef": "o1-2024-12-17_unknown",
      "numericRowCount": 4,
      "observedAtMax": "2024-12-17",
      "observedAtMin": "2024-12-17",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 79,
          "metricId": "Percent correct",
          "observedAt": "2024-12-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "32.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=70;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 466,
          "metricId": "Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0083",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=190;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0083
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 700,
          "metricId": "Score",
          "observedAt": "2024-09-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.14",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=211;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.14
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Percent correct",
        "Score"
      ],
      "modelRef": "o1-mini-2024-09-12_unknown",
      "numericRowCount": 4,
      "observedAtMax": "2024-12-22",
      "observedAtMin": "2024-09-12",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-metr_time_horizons_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 82,
          "metricId": "Percent correct",
          "observedAt": "2025-06-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "76.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=73;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 76.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2111,
          "metricId": "ECI Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "147.06",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=510;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 147.06
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2285,
          "metricId": "Overall score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "62.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=2;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 62.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Overall score",
        "Percent correct",
        "average_score"
      ],
      "modelRef": "o3-2025-04-16_unknown",
      "numericRowCount": 4,
      "observedAtMax": "2025-06-25",
      "observedAtMin": "2025-04-16",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 439,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0194",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=163;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0194
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 619,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.57",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=129;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.57
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2120,
          "metricId": "ECI Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "147.51",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=522;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 147.51
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "ECI Score",
        "Mean score",
        "Score"
      ],
      "modelRef": "o3-pro-2025-06-10_medium",
      "numericRowCount": 4,
      "observedAtMax": "2025-06-10",
      "observedAtMin": "2025-06-10",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-scicode_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1495,
          "metricId": "Accuracy",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=134;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2146,
          "metricId": "ECI Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=570;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.77
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4517,
          "metricId": "Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.423611111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=85;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.423611111111111
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score"
      ],
      "modelRef": "qwen/qwen3-235b-a22b-thinking-2507",
      "numericRowCount": 4,
      "observedAtMax": "2025-07-25",
      "observedAtMin": "2025-07-25",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 4,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external",
        "epoch-surface_evolver_bench_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1451,
          "metricId": "Accuracy",
          "observedAt": "2026-04-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00857142857142857",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=90;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.857142857142857
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4526,
          "metricId": "Score",
          "observedAt": "2026-04-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.361111111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=94;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.361111111111111
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-surface_evolver_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4810,
          "metricId": "Mean score",
          "observedAt": "2026-04-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.15625",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "surface_evolver_bench_external.csv:row=24;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.625
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Mean score",
        "Score"
      ],
      "modelRef": "trinity-large-thinking",
      "numericRowCount": 4,
      "observedAtMax": "2026-04-01",
      "observedAtMin": "2026-04-01",
      "rowCount": 4,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 28,
          "metricId": "all",
          "observedAt": "2025-12-16",
          "protocol": {
            "harness": "ML-Master 2.0",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "56.44 ± 2.47",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=16;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 56.44
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 27,
          "metricId": "high",
          "observedAt": "2025-12-16",
          "protocol": {
            "harness": "ML-Master 2.0",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "42.22 ± 2.22",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=16;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 42.22
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 25,
          "metricId": "lite",
          "observedAt": "2025-12-16",
          "protocol": {
            "harness": "ML-Master 2.0",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "75.76 ± 1.51",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=16;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 75.76
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "Deepseek-V3.2-Speciale",
      "numericRowCount": 4,
      "observedAtMax": "2025-12-16",
      "observedAtMin": "2025-12-16",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 100,
          "metricId": "all",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "7.56 ± 1.60",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=34;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 7.56
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 99,
          "metricId": "high",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "2.22 ± 2.22",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=34;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 2.22
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 97,
          "metricId": "lite",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "19.70 ± 1.52",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=34;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 19.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "claude-3-5-sonnet-20240620",
      "numericRowCount": 4,
      "observedAtMax": "2024-10-08",
      "observedAtMin": "2024-10-08",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 44,
          "metricId": "all",
          "observedAt": "2025-11-10",
          "protocol": {
            "harness": "Thesis",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "48.44 ± 3.64",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=20;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 48.44
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 43,
          "metricId": "high",
          "observedAt": "2025-11-10",
          "protocol": {
            "harness": "Thesis",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "31.11 ± 2.22",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=20;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 31.11
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 41,
          "metricId": "lite",
          "observedAt": "2025-11-10",
          "protocol": {
            "harness": "Thesis",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "65.15 ± 1.52",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=20;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 65.15
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "gpt-5-codex",
      "numericRowCount": 4,
      "observedAtMax": "2025-11-10",
      "observedAtMin": "2025-11-10",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 108,
          "metricId": "all",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "3.33 ± 0.38",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=36;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 3.33
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 107,
          "metricId": "high",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "0.00 ± 0.00",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=36;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 105,
          "metricId": "lite",
          "observedAt": "2024-10-08",
          "protocol": {
            "harness": "AIDE",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "10.23 ± 1.14",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=36;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 10.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "llama-3.1-405b-instruct",
      "numericRowCount": 4,
      "observedAtMax": "2024-10-08",
      "observedAtMin": "2024-10-08",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 76,
          "metricId": "all",
          "observedAt": "2025-05-15",
          "protocol": {
            "harness": "AIRA-dojo",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "31.60 ± 0.82",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=28;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 31.6
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 75,
          "metricId": "high",
          "observedAt": "2025-05-15",
          "protocol": {
            "harness": "AIRA-dojo",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "21.67 ± 1.07",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=28;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 21.67
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 73,
          "metricId": "lite",
          "observedAt": "2025-05-15",
          "protocol": {
            "harness": "AIRA-dojo",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "55.00 ± 1.47",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=28;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 55.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "o3",
      "numericRowCount": 4,
      "observedAtMax": "2025-05-15",
      "observedAtMin": "2025-05-15",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 80,
          "metricId": "all",
          "observedAt": "2025-08-15",
          "protocol": {
            "harness": "R&D-Agent",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "30.22 ± 0.89",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=29;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 30.22
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 79,
          "metricId": "high",
          "observedAt": "2025-08-15",
          "protocol": {
            "harness": "R&D-Agent",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "26.67 ± 0.00",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=29;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 26.67
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 77,
          "metricId": "lite",
          "observedAt": "2025-08-15",
          "protocol": {
            "harness": "R&D-Agent",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "51.52 ± 4.01",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=29;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 51.52
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "o3 + GPT-4.1",
      "numericRowCount": 4,
      "observedAtMax": "2025-08-15",
      "observedAtMin": "2025-08-15",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mle-bench"
      ],
      "examples": [
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 72,
          "metricId": "all",
          "observedAt": "2025-07-28",
          "protocol": {
            "harness": "Neo multi-agent",
            "split": "all",
            "subject_type": "system"
          },
          "rawValue": "34.22 ± 0.89",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=27;column=All (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 34.22
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 71,
          "metricId": "high",
          "observedAt": "2025-07-28",
          "protocol": {
            "harness": "Neo multi-agent",
            "split": "high",
            "subject_type": "system"
          },
          "rawValue": "24.44 ± 2.22",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=27;column=High (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 24.44
        },
        {
          "artifact": "src-mle-bench/candidates.jsonl",
          "benchmarkId": "mle-bench",
          "benchmarkName": "MLE-bench",
          "evidenceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "line": 69,
          "metricId": "lite",
          "observedAt": "2025-07-28",
          "protocol": {
            "harness": "Neo multi-agent",
            "split": "lite",
            "subject_type": "system"
          },
          "rawValue": "48.48 ± 1.52",
          "sourceId": "src-mle-bench",
          "sourceLabel": "OpenAI · MLE-bench",
          "sourceLocator": "README.md:row=27;column=Low == Lite (%)",
          "sourceUrl": "https://raw.githubusercontent.com/openai/mle-bench/main/README.md",
          "unit": "percent",
          "value": 48.48
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "all",
        "high",
        "lite",
        "medium"
      ],
      "modelRef": "undisclosed",
      "numericRowCount": 4,
      "observedAtMax": "2025-07-28",
      "observedAtMin": "2025-07-28",
      "rowCount": 4,
      "sourceId": "src-mle-bench",
      "sourceIds": [
        "src-mle-bench"
      ],
      "sourceLabels": [
        "OpenAI · MLE-bench"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/openai/mle-bench/main/README.md"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 329,
          "metricId": "resolved",
          "observedAt": "2025-03-06",
          "protocol": {
            "checked": true,
            "harness": "SWE-Fixer",
            "leaderboard_variant": "Lite",
            "scaffold": "SWE-Fixer",
            "subject_type": "model"
          },
          "rawValue": 24.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[64]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 24.67
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 333,
          "metricId": "resolved",
          "observedAt": "2024-11-28",
          "protocol": {
            "checked": false,
            "harness": "SWE-Fixer",
            "leaderboard_variant": "Lite",
            "scaffold": "SWE-Fixer",
            "subject_type": "model"
          },
          "rawValue": 23.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[68]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 23.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 233,
          "metricId": "resolved",
          "observedAt": "2025-03-06",
          "protocol": {
            "checked": true,
            "harness": "SWE-Fixer",
            "leaderboard_variant": "Verified",
            "scaffold": "SWE-Fixer",
            "subject_type": "model"
          },
          "rawValue": 32.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[148]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 32.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Qwen2.5 (7B + 72B)",
      "numericRowCount": 4,
      "observedAtMax": "2025-03-06",
      "observedAtMin": "2024-11-28",
      "rowCount": 4,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-multimodal",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 27,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 58.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[26]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 58.4
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 349,
          "metricId": "resolved",
          "observedAt": "2025-07-01",
          "protocol": {
            "checked": true,
            "harness": "GUIRepair",
            "leaderboard_variant": "Multimodal",
            "scaffold": "GUIRepair",
            "subject_type": "system"
          },
          "rawValue": 35.98,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[0]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 35.98
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 350,
          "metricId": "resolved",
          "observedAt": "2025-11-17",
          "protocol": {
            "checked": true,
            "harness": "Codefuse_Pycfuse_SVR",
            "leaderboard_variant": "Multimodal",
            "scaffold": "Codefuse_Pycfuse_SVR",
            "subject_type": "system"
          },
          "rawValue": 35.98,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[1]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 35.98
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "o3",
      "numericRowCount": 4,
      "observedAtMax": "2025-11-17",
      "observedAtMin": "2025-07-01",
      "rowCount": 4,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-multimodal",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 36,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 45.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[35]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 45.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 353,
          "metricId": "resolved",
          "observedAt": "2025-05-31",
          "protocol": {
            "checked": true,
            "harness": "GUIRepair",
            "leaderboard_variant": "Multimodal",
            "scaffold": "GUIRepair",
            "subject_type": "system"
          },
          "rawValue": 33.85,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[4]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 33.85
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 207,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 45.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[122]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 45.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 4
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "o4-mini",
      "numericRowCount": 4,
      "observedAtMax": "2025-07-26",
      "observedAtMin": "2025-05-03",
      "rowCount": 4,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 4
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 55,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.47,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=55;row=54",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 60.47
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 61,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=61;row=60",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 59.2
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 68,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.73,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=68;row=67",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "RedHatAI/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 58,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.47,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=58;row=57",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 60.47
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 64,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=64;row=63",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 59.2
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 71,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.73,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=71;row=70",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-MXFP4_MOE-dequant-bf16-vllm",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 56,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.47,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=56;row=55",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 60.47
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 62,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=62;row=61",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 59.2
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 69,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.73,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=69;row=68",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q3_K_M-dequant-bf16-vllm",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 57,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.47,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=57;row=56",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 60.47
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 63,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=63;row=62",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 59.2
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 70,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.73,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=70;row=69",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q4_K_M-dequant-bf16-vllm",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 54,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.47,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=54;row=53",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 60.47
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 60,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=60;row=59",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 59.2
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 67,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.73,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=67;row=66",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 51,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified_high.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 62.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=51;row=50",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 62.4
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 74,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified_medium.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=74;row=73",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 52.6
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 76,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified_low.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 47.9,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=76;row=75",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 47.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-120b",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 53,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified_high.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.7,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=53;row=52",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 60.7
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 72,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified_medium.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=72;row=71",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.2
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2508.10925",
          "line": 77,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified_low.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=77;row=76",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 37.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-20b",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2090,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1255.4301815248325,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=102",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1255.4301815248325
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2252,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1255.4301815248325,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=264",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1255.4301815248325
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2372,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-html",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1265.3165957420365,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=384",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1265.3165957420365
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "KAT-Coder-Pro-V1",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 145,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1394.6252116734638,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=144",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1394.6252116734638
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 548,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1404.948939865232,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=547",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1404.948939865232
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 897,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1413.0911207881393,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=896",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1413.0911207881393
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-experimental-chat-10-20",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 115,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1416.2235896291534,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=114",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1416.2235896291534
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 474,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1468.328775491344,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=473",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1468.328775491344
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 873,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1432.1131953984464,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=872",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1432.1131953984464
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-experimental-chat-11-10",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 509,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1450.438233587176,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=508",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1450.438233587176
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 878,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1428.2977014798255,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=877",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1428.2977014798255
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 100,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1419.8797027691078,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=99",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1419.8797027691078
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-experimental-chat-12-10",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 44,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1447.3283135763743,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=43",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1447.3283135763743
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 494,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1457.3810665362882,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=493",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1457.3810665362882
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 794,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1484.0804460624972,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=793",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1484.0804460624972
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-experimental-chat-26-02-10",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 233,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1291.5789095040254,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=232",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1291.5789095040254
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 611,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1300.9396764117669,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=610",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1300.9396764117669
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 984,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1312.6844152355077,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=983",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1312.6844152355077
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "athene-v2-chat",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 107,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1418.1493955945834,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=106",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1418.1493955945834
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 540,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1419.6768917294798,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=539",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1419.6768917294798
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 801,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1479.592275269696,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=800",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1479.592275269696
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-opus-4-1-20250805-thinking-16k",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 205,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1331.1822505166074,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=204",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1331.1822505166074
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 602,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1322.1951039437,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=601",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1322.1951039437
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 971,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1331.0079797567585,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=970",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1331.0079797567585
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "command-a-03-2025",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 163,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1372.7410695141282,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=162",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1372.7410695141282
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 552,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1399.8006944105045,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=551",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1399.8006944105045
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 939,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1371.9058015381122,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=938",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1371.9058015381122
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-r1",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 258,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1271.3775893078446,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=257",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1271.3775893078446
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 618,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1280.320884936974,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=617",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1280.320884936974
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 994,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1301.845085498423,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=993",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1301.845085498423
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-v2.5",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 232,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1294.056683741669,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=231",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1294.056683741669
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 605,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1316.5307118617561,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=604",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1316.5307118617561
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 987,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1309.9391367084556,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=986",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1309.9391367084556
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-v2.5-1210",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 203,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1332.6019590672786,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=202",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1332.6019590672786
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 593,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1337.5895101439542,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=592",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1337.5895101439542
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 974,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1325.4927482947976,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=973",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1325.4927482947976
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-v3",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 159,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1375.0452821722458,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=158",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1375.0452821722458
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 559,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1387.113117107074,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=558",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1387.113117107074
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 941,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1369.3972201749214,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=940",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1369.3972201749214
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-v3-0324",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2105,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1080.1083205339569,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=117",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1080.1083205339569
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2267,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1080.1083205339569,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=279",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1080.1083205339569
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2384,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-html",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1087.0479173687204,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=396",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1087.0479173687204
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "devstral-medium-2507",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 440,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1496.4921438677047,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=439",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1496.4921438677047
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 52,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.4736265262584,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=51",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.4736265262584
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 834,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1454.6355639565081,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=833",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1454.6355639565081
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ernie-5.0-0110",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 434,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1501.6206002476083,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=433",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1501.6206002476083
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 81,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1429.0993880432568,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=80",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1429.0993880432568
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 898,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1412.9237804111199,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=897",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1412.9237804111199
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ernie-5.0-preview-1022",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 456,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1480.1075887130432,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=455",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1480.1075887130432
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 55,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1442.4009452079038,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=54",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1442.4009452079038
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 881,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1425.9058441099417,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=880",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1425.9058441099417
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ernie-5.0-preview-1203",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 236,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1289.7790761610656,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=235",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1289.7790761610656
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 615,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1294.899490342698,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=614",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1294.899490342698
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1000,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1290.9744018711885,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=999",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1290.9744018711885
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4-plus",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 471,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1469.6061035922062,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=470",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1469.6061035922062
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 80,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1429.4159275769518,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=79",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1429.4159275769518
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 870,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1434.1429032121005,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=869",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1434.1429032121005
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4.5",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 151,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1382.8005392208363,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=150",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1382.8005392208363
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 528,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1429.4741220145954,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=527",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1429.4741220145954
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 917,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1396.46567158266,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=916",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1396.46567158266
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4.5-air",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 149,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1387.5600041070174,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=148",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1387.5600041070174
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 533,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1424.7187149587353,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=532",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1424.7187149587353
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 907,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1405.9348645609666,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=906",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1405.9348645609666
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-5.3-chat-latest",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 172,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1365.6054393336299,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=171",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1365.6054393336299
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 569,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1377.3447973329403,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=568",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1377.3447973329403
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 928,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1380.5278516460353,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=927",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1380.5278516460353
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-oss-120b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 238,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1287.7686931591534,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=237",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1287.7686931591534
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 609,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1307.763566519564,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=608",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1307.763566519564
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 988,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1308.4100617804668,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=987",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1308.4100617804668
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-oss-20b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 176,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1362.645814211086,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=175",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1362.645814211086
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 571,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1375.8602170138909,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=570",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1375.8602170138909
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 948,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1366.1265359235124,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=947",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1366.1265359235124
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-3-mini-beta",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 168,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1366.9325595043238,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=167",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1366.9325595043238
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 565,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.3727518241035,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=564",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.3727518241035
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 936,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1376.234461941551,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=935",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1376.234461941551
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-3-mini-high",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 513,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1447.7735247338471,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=512",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1447.7735247338471
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 88,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1425.1060926897815,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=87",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1425.1060926897815
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 872,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1432.4254904657028,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=871",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1432.4254904657028
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-3-preview-02-24",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 126,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1408.4235624886119,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=125",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1408.4235624886119
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 472,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1469.50656729937,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=471",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1469.50656729937
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 875,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1429.5843164693222,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=874",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1429.5843164693222
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4-fast-chat",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 466,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1471.9975230678156,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=465",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1471.9975230678156
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 74,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1436.4305767337119,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=73",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1436.4305767337119
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 853,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.2928029058016,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=852",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.2928029058016
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4.1",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-webdev"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2103,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1164.5131317022435,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=115",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1164.5131317022435
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2265,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1164.5131317022435,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=277",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1164.5131317022435
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-webdev",
          "benchmarkName": "Arena WebDev",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2382,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "webdev",
            "category": "webdev-html",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1172.1423271742442,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=webdev;split=latest;row_idx=394",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1172.1423271742442
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-code-fast-1",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 237,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1288.0450672005713,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=236",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1288.0450672005713
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 589,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1345.4948090541707,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=588",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1345.4948090541707
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 989,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1307.2394463195815,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=988",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1307.2394463195815
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-large-2025-02-10",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 137,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1399.7588197871698,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=136",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1399.7588197871698
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 530,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1428.14075514464,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=529",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1428.14075514464
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 923,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1388.3527233503794,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=922",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1388.3527233503794
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-t1-20250711",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 220,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1311.5095095348074,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=219",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1311.5095095348074
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 579,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1357.1847501209768,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=578",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1357.1847501209768
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 978,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1319.869560155202,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=977",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1319.869560155202
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-turbo-0110",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 213,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1319.820643908577,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=212",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1319.820643908577
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 600,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1323.0907462560058,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=599",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1323.0907462560058
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 965,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1343.0761772167996,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=964",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1343.0761772167996
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-turbos-20250226",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 156,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1376.4186845245963,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=155",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1376.4186845245963
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 542,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1418.2241151937958,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=541",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1418.2241151937958
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 955,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1361.171969546864,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=954",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1361.171969546864
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-turbos-20250416",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 182,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1355.7243262585632,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=181",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1355.7243262585632
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 580,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1357.0966916879838,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=579",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1357.0966916879838
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 943,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1368.1932275290135,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=942",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1368.1932275290135
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "intellect-3",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1128,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1004.7879384391505,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=127",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1004.7879384391505
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1271,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1059.7524705110347,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=270",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1059.7524705110347
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1637,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1038.3600895941115,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=636",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1038.3600895941115
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "internvl2-26b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1140,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 941.536149160439,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=139",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 941.536149160439
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1283,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 945.2327143680229,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=282",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 945.2327143680229
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1650,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 986.9456204496441,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=649",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 986.9456204496441
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "internvl2-4b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 164,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1370.8932249056738,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=163",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1370.8932249056738
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 553,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1397.9732801567136,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=552",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1397.9732801567136
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 935,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1377.9503400597134,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=934",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1377.9503400597134
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "kimi-k2-0711-preview",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 154,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.0607449179588,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=153",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.0607449179588
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 545,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1413.4113893640688,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=544",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1413.4113893640688
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 912,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1399.972592019664,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=911",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1399.972592019664
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "kimi-k2-0905-preview",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 175,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1363.0243005399088,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=174",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1363.0243005399088
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 546,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1410.4738649092,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=545",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1410.4738649092
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 922,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1390.332950132382,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=921",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1390.332950132382
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ling-flash-2.0",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 243,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1283.795309432649,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=242",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1283.795309432649
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 642,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1241.245784333175,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=641",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1241.245784333175
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 999,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1291.8431501458917,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=998",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1291.8431501458917
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-405b-instruct-bf16",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1137,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 966.6000649030858,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=136",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 966.6000649030858
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1287,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 909.8442728648067,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=286",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 909.8442728648067
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1648,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 998.0261096555652,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=647",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 998.0261096555652
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.2-vision-11b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1129,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 999.8796597284471,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=128",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 999.8796597284471
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1281,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 947.3330684732437,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=280",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 947.3330684732437
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1642,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1014.2217205820737,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=641",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1014.2217205820737
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.2-vision-90b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 222,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1308.4603186362888,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=221",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1308.4603186362888
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 619,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1276.4003348121623,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=618",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1276.4003348121623
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 996,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1296.1330131900572,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=995",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1296.1330131900572
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.3-nemotron-49b-super-v1",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1142,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 936.6672265877847,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=141",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 936.6672265877847
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1288,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 909.6666554005448,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=287",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 909.6666554005448
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1652,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 971.4525746606095,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=651",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 971.4525746606095
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llava-v1.6-34b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 525,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1432.0211787594658,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=524",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1432.0211787594658
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 818,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1467.8490926900636,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=817",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1467.8490926900636
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 96,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1422.217999442223,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=95",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1422.217999442223
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "longcat-flash-chat",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 478,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1466.6722647410181,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=477",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1466.6722647410181
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 810,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1472.9946696913476,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=809",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1472.9946696913476
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 87,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1425.7601015834514,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=86",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1425.7601015834514
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "longcat-flash-chat-2602-exp",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 272,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1254.3846857299327,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=271",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1254.3846857299327
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 659,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1222.0657245946927,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=658",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1222.0657245946927
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 979,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1319.5786020835997,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=978",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1319.5786020835997
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "magistral-medium-2506",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 191,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1341.910642817768,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=190",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1341.910642817768
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 576,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1358.734537531799,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=575",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1358.734537531799
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 957,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1358.6337527365379,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=956",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1358.6337527365379
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "minimax-m1",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1125,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1012.5161131639651,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=124",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1012.5161131639651
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1285,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 931.5032908739345,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=284",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 931.5032908739345
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1636,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1046.2748993434157,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=635",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1046.2748993434157
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "molmo-72b-0924",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1138,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 958.919061013075,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=137",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 958.919061013075
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1289,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 885.1906450381246,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=288",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 885.1906450381246
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1643,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1013.7060059955879,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=642",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1013.7060059955879
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "molmo-7b-d-0924",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 177,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1362.1681415094408,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=176",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1362.1681415094408
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 575,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1370.9224686271286,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=574",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1370.9224686271286
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 924,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1387.2212852703258,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=923",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1387.2212852703258
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nova-2-lite",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 188,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1348.9162718918994,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=187",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1348.9162718918994
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 555,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1391.5098880454213,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=554",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1391.5098880454213
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 931,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.5080614203152,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=930",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.5080614203152
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nvidia-nemotron-3-nano-30b-a3b-bf16",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 155,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1377.6114395025545,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=154",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1377.6114395025545
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 531,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1426.7028041043475,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=530",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1426.7028041043475
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 908,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1404.6653445008335,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=907",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1404.6653445008335
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nvidia-nemotron-3-super-120b-a12b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 443,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1493.6832415013948,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=442",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1493.6832415013948
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 51,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.9521839587653,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=50",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.9521839587653
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 815,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1468.792636468878,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=814",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1468.792636468878
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nvidia-nemotron-3-ultra-550b-a55b-nvfp4",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 217,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1317.1950639943689,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=216",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1317.1950639943689
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 606,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1313.227726624125,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=605",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1313.227726624125
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 951,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1362.8249038750585,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=950",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1362.8249038750585
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o1-mini",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 185,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1352.9005377977603,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=184",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1352.9005377977603
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 598,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1324.537401498479,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=597",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1324.537401498479
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 947,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1367.1512958623262,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=946",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1367.1512958623262
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o1-preview",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 216,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1319.1641123038994,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=215",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1319.1641123038994
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 601,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1322.8489594387916,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=600",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1322.8489594387916
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 953,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1362.6775246557518,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=952",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1362.6775246557518
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o3-mini",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 199,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1336.637800941528,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=198",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1336.637800941528
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 568,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1378.1009966637473,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=567",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1378.1009966637473
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 933,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1378.6849762825175,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=932",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1378.6849762825175
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o3-mini-high",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 229,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1298.5533904800536,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=228",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1298.5533904800536
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 614,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1298.0244164335052,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=613",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1298.0244164335052
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 977,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1321.3116817420391,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=976",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1321.3116817420391
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "olmo-3-32b-think",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 221,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1311.0330933391137,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=220",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1311.0330933391137
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 612,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1300.7311808144057,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=611",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1300.7311808144057
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 964,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1347.426037342718,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=963",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1347.426037342718
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "olmo-3.1-32b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1126,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1008.7972471743793,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=125",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1008.7972471743793
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1282,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 945.6285175520654,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=281",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 945.6285175520654
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1639,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1027.8196674804512,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=638",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1027.8196674804512
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "pixtral-12b-2409",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1114,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1089.6556911499433,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=113",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1089.6556911499433
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1268,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1077.7911348012315,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=267",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1077.7911348012315
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1625,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1101.1021169315904,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=624",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1101.1021169315904
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "pixtral-large-2411",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 210,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1326.5042820394744,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=209",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1326.5042820394744
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 588,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1346.4909222733284,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=587",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1346.4909222733284
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 972,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1328.5670524374873,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=971",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1328.5670524374873
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen-plus-0125",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1121,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1047.154757195472,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=120",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1047.154757195472
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1270,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1066.71231832016,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=269",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1066.71231832016
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1635,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1054.8718899030316,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=634",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1054.8718899030316
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2-vl-72b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1133,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 990.1283898740592,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=132",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 990.1283898740592
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1277,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 994.3249686411045,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=276",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 994.3249686411045
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1645,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1006.9984125282953,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=644",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1006.9984125282953
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2-vl-7b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 260,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1269.1439358282387,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=259",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1269.1439358282387
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 625,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1271.4121678262445,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=624",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1271.4121678262445
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 998,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1292.9622960014003,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=997",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1292.9622960014003
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2.5-72b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 169,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1366.6946750922098,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=168",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1366.6946750922098
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 564,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.5049099154996,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=563",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.5049099154996
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 956,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1360.02480400183,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=955",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1360.02480400183
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2.5-max",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 228,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1299.2667659228027,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=227",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1299.2667659228027
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 607,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1312.9092270041624,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=606",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1312.9092270041624
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 981,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1314.7517449306415,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=980",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1314.7517449306415
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2.5-plus-1127",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1110,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1107.6424562039888,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=109",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1107.6424562039888
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1266,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1101.6972198243145,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=265",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1101.6972198243145
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1623,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1115.9225677751724,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=622",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1115.9225677751724
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2.5-vl-72b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 170,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1365.9570774626795,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=169",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1365.9570774626795
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 566,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.192913314635,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=565",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.192913314635
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 926,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1385.7612056945788,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=925",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1385.7612056945788
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-235b-a22b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 104,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1419.3461305293924,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=103",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1419.3461305293924
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 482,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1465.9241721449014,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=481",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1465.9241721449014
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 850,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1444.7937341337595,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=849",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1444.7937341337595
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-235b-a22b-instruct-2507",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 146,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1393.41868101409,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=145",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1393.41868101409
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 536,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1424.391666208244,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=535",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1424.391666208244
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 915,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1398.4755812836981,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=914",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1398.4755812836981
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-235b-a22b-no-thinking",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 118,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1413.7929201461397,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=117",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1413.7929201461397
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 465,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1473.2018809043332,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=464",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1473.2018809043332
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 884,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1424.580195860237,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=883",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1424.580195860237
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-235b-a22b-thinking-2507",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 218,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1316.8556700247236,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=217",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1316.8556700247236
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 582,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1354.5926374455562,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=581",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1354.5926374455562
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 969,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1337.7567177244546,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=968",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1337.7567177244546
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-30b-a3b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 150,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1384.3208285247435,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=149",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1384.3208285247435
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 522,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1432.7622709375935,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=521",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1432.7622709375935
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 891,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1417.87264419656,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=890",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1417.87264419656
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-30b-a3b-instruct-2507",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 194,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1340.0813859727234,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=193",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1340.0813859727234
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 578,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1357.2091344441774,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=577",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1357.2091344441774
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 959,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1358.163837654478,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=958",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1358.163837654478
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-32b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 119,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1412.7255837015796,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=118",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1412.7255837015796
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 532,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1425.759286170308,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=531",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1425.759286170308
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 861,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1438.7502037241456,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=860",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1438.7502037241456
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-max-2025-09-23",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 447,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1489.996843791813,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=446",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1489.996843791813
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 65,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1439.073379959434,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=64",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1439.073379959434
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 829,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1456.9033254148662,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=828",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1456.9033254148662
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-max-preview",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 106,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1418.5810951013584,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=105",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1418.5810951013584
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 470,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1470.0764658692483,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=469",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1470.0764658692483
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 856,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1441.5509919371789,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=855",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1441.5509919371789
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-next-80b-a3b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 167,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1367.5096728133117,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=166",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1367.5096728133117
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 543,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1417.3558797232147,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=542",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1417.3558797232147
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 920,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1390.941251316613,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=919",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1390.941251316613
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen3-next-80b-a3b-thinking",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 208,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1329.0593086656377,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=207",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1329.0593086656377
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 570,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1376.3483899402513,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=569",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1376.3483899402513
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 970,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1334.355516659401,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=969",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1334.355516659401
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwq-32b",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 204,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1331.5278378519874,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=203",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1331.5278378519874
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 547,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1406.3561208424226,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=546",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1406.3561208424226
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 950,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1365.8672149349125,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=949",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1365.8672149349125
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ring-flash-2.0",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 211,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1321.027314890262,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=210",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1321.027314890262
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 595,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1330.1075330845251,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=594",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1330.1075330845251
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 980,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1316.7192788058253,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=979",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1316.7192788058253
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "step-2-16k-exp-202412",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 134,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1403.8195230615129,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=133",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1403.8195230615129
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 511,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1449.9266049406874,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=510",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1449.9266049406874
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 864,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1436.609379966051,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=863",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1436.609379966051
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "step-3.5-flash",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 195,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1338.6639182470606,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=194",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1338.6639182470606
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 583,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1353.2683677451812,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=582",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1353.2683677451812
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 929,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1380.1836355944888,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=928",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1380.1836355944888
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "trinity-large-preview",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 225,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1301.6570633223864,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=224",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1301.6570633223864
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 603,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1318.8878880960483,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=602",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1318.8878880960483
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 982,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1312.8846952177767,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=981",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1312.8846952177767
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "yi-lightning",
      "numericRowCount": 3,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 3,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-gsm8k_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 930,
          "metricId": "Average",
          "observedAt": "2023-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4301",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=37;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4301
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 2993,
          "metricId": "EM",
          "observedAt": "2023-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.268",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=97;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 26.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2309.10305",
          "line": 3642,
          "metricId": "EM",
          "observedAt": "2023-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.516",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=9;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 51.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Average",
        "EM"
      ],
      "modelRef": "Baichuan-13B-Base",
      "numericRowCount": 3,
      "observedAtMax": "2023-07-11",
      "observedAtMin": "2023-07-11",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-cursorbench_external",
        "epoch-frontiercode_external",
        "epoch-frontierswe_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-cursorbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1554,
          "metricId": "Score",
          "observedAt": "2026-05-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.561",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "cursorbench_external.csv:row=54;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.561
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiercode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2384,
          "metricId": "Main score",
          "observedAt": "2026-05-18",
          "protocol": {
            "harness": "cursor-cli",
            "subject_type": "system"
          },
          "rawValue": "0.2564",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiercode_external.csv:row=21;column=Main score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.2564
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontierswe_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2581,
          "metricId": "Implementation rank",
          "observedAt": "2026-05-18",
          "protocol": {
            "harness": "Cursor CLI",
            "subject_type": "system"
          },
          "rawValue": "10.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontierswe_external.csv:row=16;column=Implementation rank",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "rank",
          "value": 10.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Implementation rank",
        "Main score",
        "Score"
      ],
      "modelRef": "Composer 2.5",
      "numericRowCount": 3,
      "observedAtMax": "2026-05-18",
      "observedAtMin": "2026-05-18",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_ai2_external",
        "epoch-gsm8k_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_ai2_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 831,
          "metricId": "Challenge score",
          "observedAt": "2024-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.643",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_ai2_external.csv:row=106;column=Challenge score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 64.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3082,
          "metricId": "EM",
          "observedAt": "2024-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.858",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=205;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 85.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-wino_grande_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2409.12186",
          "line": 5621,
          "metricId": "Accuracy",
          "observedAt": "2024-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.837",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "wino_grande_external.csv:row=63;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 83.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Challenge score",
        "EM"
      ],
      "modelRef": "DeepSeek-Coder-V2-Base",
      "numericRowCount": 3,
      "observedAtMax": "2024-06-17",
      "observedAtMin": "2024-06-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-forecastbench_external",
        "epoch-gso_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2328,
          "metricId": "Overall score",
          "observedAt": "2025-07-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=45;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gso_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3145,
          "metricId": "Score OPT@1",
          "observedAt": "2025-07-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0294",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gso_external.csv:row=36;column=Score OPT@1",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0294
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3293,
          "metricId": "Accuracy",
          "observedAt": "2025-07-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0812",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hle_external.csv:row=33;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Overall score",
        "Score OPT@1"
      ],
      "modelRef": "GLM-4.5-Air",
      "numericRowCount": 3,
      "observedAtMax": "2025-07-20",
      "observedAtMin": "2025-07-20",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3546,
          "metricId": "mean_score",
          "observedAt": "2024-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.22686933534743203",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=79;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 22.686933534743204
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4192,
          "metricId": "mean_score",
          "observedAt": "2024-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.025",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=185;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2915,
          "metricId": "mean_score",
          "observedAt": "2024-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3746843434343434",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=241;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 37.46843434343434
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "Hermes-2-Theta-Llama-3-70B",
      "numericRowCount": 3,
      "observedAtMax": "2024-06-20",
      "observedAtMin": "2024-06-20",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3571,
          "metricId": "mean_score",
          "observedAt": "2024-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.426642749244713",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=104;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.6642749244713
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4198,
          "metricId": "mean_score",
          "observedAt": "2024-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.044444444444444446",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=191;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.444444444444445
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2935,
          "metricId": "mean_score",
          "observedAt": "2024-11-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.46275252525252525",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=261;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.275252525252526
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "Llama-3.1-Tulu-3-70B-DPO",
      "numericRowCount": 3,
      "observedAtMax": "2024-11-21",
      "observedAtMin": "2024-11-21",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2159,
          "metricId": "ECI Score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "122.56",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=607;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 122.56
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2351,
          "metricId": "Overall score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=68;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-wino_grande_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2407.21783",
          "line": 5605,
          "metricId": "Accuracy",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.835",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "wino_grande_external.csv:row=47;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 83.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score"
      ],
      "modelRef": "Meta-Llama-3-70B",
      "numericRowCount": 3,
      "observedAtMax": "2024-04-18",
      "observedAtMin": "2024-04-18",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2162,
          "metricId": "ECI Score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "116.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=611;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 116.32
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2363,
          "metricId": "Overall score",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "52.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=80;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-wino_grande_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/2407.21783",
          "line": 5604,
          "metricId": "Accuracy",
          "observedAt": "2024-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.757",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "wino_grande_external.csv:row=46;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score"
      ],
      "modelRef": "Meta-Llama-3-8B",
      "numericRowCount": 3,
      "observedAtMax": "2024-04-18",
      "observedAtMin": "2024-04-18",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-terminalbench_external",
        "epoch-vending_bench_2_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4980,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-01",
          "protocol": {
            "harness": "Terminus 2",
            "subject_type": "system"
          },
          "rawValue": "0.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=136;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 30.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-vending_bench_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5232,
          "metricId": "Score",
          "observedAt": "2025-10-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "160.5959999999999",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "vending_bench_2_external.csv:row=53;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 160.5959999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-webdev_arena_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5378,
          "metricId": "Arena Score",
          "observedAt": "2025-10-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1297.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "webdev_arena_external.csv:row=104;column=Arena Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "elo",
          "value": 1297.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy mean",
        "Arena Score",
        "Score"
      ],
      "modelRef": "MiniMax-M2",
      "numericRowCount": 3,
      "observedAtMax": "2025-11-01",
      "observedAtMin": "2025-10-27",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-mmlu_external",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3556,
          "metricId": "mean_score",
          "observedAt": "2024-05-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.03597054380664653",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=89;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.5970543806646527
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3784,
          "metricId": "EM",
          "observedAt": "2024-05-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.599",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=164;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2919,
          "metricId": "mean_score",
          "observedAt": "2024-05-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1518308080808081",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=245;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.18308080808081
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "mean_score"
      ],
      "modelRef": "Mistral-7B-Instruct-v0.3",
      "numericRowCount": 3,
      "observedAtMax": "2024-05-27",
      "observedAtMin": "2024-05-27",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2160,
          "metricId": "ECI Score",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "121.22",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=609;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 121.22
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2356,
          "metricId": "Overall score",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "56.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=73;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5559,
          "metricId": "Accuracy",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0317",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=167;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall score"
      ],
      "modelRef": "Mixtral-8x22B-Instruct-v0.1",
      "numericRowCount": 3,
      "observedAtMax": "2024-04-17",
      "observedAtMin": "2024-04-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1061,
          "metricId": "Score",
          "observedAt": "2024-08-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.846",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=121;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.846
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 2996,
          "metricId": "EM",
          "observedAt": "2024-08-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.887",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=100;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 88.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-piqa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4276,
          "metricId": "Score",
          "observedAt": "2024-08-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.886",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "piqa_external.csv:row=53;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.886
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Score"
      ],
      "modelRef": "Phi-3.5-MoE-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2024-08-17",
      "observedAtMin": "2024-08-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 1060,
          "metricId": "Score",
          "observedAt": "2024-08-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.78",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=120;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.78
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 2995,
          "metricId": "EM",
          "observedAt": "2024-08-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.862",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=99;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 86.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-piqa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4275,
          "metricId": "Score",
          "observedAt": "2024-08-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.81",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "piqa_external.csv:row=52;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.81
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Score"
      ],
      "modelRef": "Phi-3.5-mini-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2024-08-16",
      "observedAtMin": "2024-08-16",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "frontiermath"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1922,
          "metricId": "ECI Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=215;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.77
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2531,
          "metricId": "mean_score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=38;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "frontiermath",
          "benchmarkName": "FrontierMath T1–T3",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2428,
          "metricId": "mean_score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.08480565371024736",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath.csv:row=36;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.480565371024735
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "Qwen/Qwen3-235B-A22B-Thinking-2507",
      "numericRowCount": 3,
      "observedAtMax": "2025-07-25",
      "observedAtMin": "2025-07-25",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 38,
          "metricId": "Percent correct",
          "observedAt": "2025-05-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "40.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=25;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1459,
          "metricId": "Accuracy",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00285714285714286",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=98;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.28571428571428603
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4478,
          "metricId": "Score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.354166666666667",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=46;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.354166666666667
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Percent correct",
        "Score"
      ],
      "modelRef": "Qwen3-32B",
      "numericRowCount": 3,
      "observedAtMax": "2025-05-08",
      "observedAtMin": "2025-04-29",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 180,
          "metricId": "Performance",
          "observedAt": "2026-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "432.57",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=94;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 432.57
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1423,
          "metricId": "Accuracy",
          "observedAt": "2026-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0371428571428571",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=62;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.7142857142857104
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4516,
          "metricId": "Score",
          "observedAt": "2026-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.423611111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=84;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.423611111111111
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Performance",
        "Score"
      ],
      "modelRef": "Ring-2.6-1T",
      "numericRowCount": 3,
      "observedAtMax": "2026-05-14",
      "observedAtMin": "2026-05-14",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-common_sense_qa_2_external",
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 1142,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.899",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=206;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.899
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-common_sense_qa_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/abs/2305.05921",
          "line": 1360,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.602",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "common_sense_qa_2_external.csv:row=8;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.602
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 4786,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.864",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=11;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.864
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "T5-3B",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 1140,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.814",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=204;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.814
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2101.03961",
          "line": 4779,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.751",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=3;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.751
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 4784,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.762",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=9;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.762
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "T5-Base",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1054,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.581",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=87;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.581
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2965,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.006",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=64;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3149,
          "metricId": "Overall accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.435",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=2;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "ada",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1055,
          "metricId": "Score",
          "observedAt": "2020-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.574",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=88;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.574
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2964,
          "metricId": "EM",
          "observedAt": "2020-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.007",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=63;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.7000000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3150,
          "metricId": "Overall accuracy",
          "observedAt": "2020-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.555",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=4;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 55.50000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "babbage",
      "numericRowCount": 3,
      "observedAtMax": "2020-06-22",
      "observedAtMin": "2020-06-22",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1050,
          "metricId": "Score",
          "observedAt": "2022-07-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.704",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=72;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.704
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2957,
          "metricId": "EM",
          "observedAt": "2022-07-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.095",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=29;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3153,
          "metricId": "Overall accuracy",
          "observedAt": "2022-07-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.744",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=7;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 74.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "bloom",
      "numericRowCount": 3,
      "observedAtMax": "2022-07-06",
      "observedAtMin": "2022-07-06",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-mmlu_external",
        "epoch-simplebench_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3667,
          "metricId": "EM",
          "observedAt": "2024-08-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.694",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=36;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 69.39999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-simplebench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4689,
          "metricId": "Score (AVG@5)",
          "observedAt": "2024-08-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.174",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "simplebench_external.csv:row=104;column=Score (AVG@5)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.174
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3460,
          "metricId": "Global average",
          "observedAt": "2024-08-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "31.76",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=56;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 31.76
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Global average",
        "Score (AVG@5)"
      ],
      "modelRef": "c4ai-command-r-plus-08-2024",
      "numericRowCount": 3,
      "observedAtMax": "2024-08-30",
      "observedAtMin": "2024-08-30",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 475,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.004",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=199;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.004
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 707,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.116",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=218;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.116
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2206,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=660;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_1K",
      "numericRowCount": 3,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 461,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.009",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=185;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.009
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 685,
          "metricId": "Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.212",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=196;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.212
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2199,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=653;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_8K",
      "numericRowCount": 3,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 463,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0085",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=187;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0085
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 670,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.28",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=181;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.28
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2195,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=647;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "claude-sonnet-4-20250514_1K",
      "numericRowCount": 3,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1943,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=238;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4170,
          "metricId": "mean_score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6888888888888889",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=163;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.88888888888889
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2838,
          "metricId": "mean_score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.7781991873045173",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=164;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.81991873045173
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "claude-sonnet-4-20250514_59K",
      "numericRowCount": 3,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 433,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0212",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=157;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0212
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 667,
          "metricId": "Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=178;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.29
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2194,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=646;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "claude-sonnet-4-20250514_8K",
      "numericRowCount": 3,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "osworld"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "osworld",
          "benchmarkName": "OSWorld 2.0",
          "evidenceUrl": "https://os-world.github.io/",
          "line": 3987,
          "metricId": "Score",
          "observedAt": "2025-03-11",
          "protocol": {
            "harness": "computer-use-preview (50 steps)",
            "subject_type": "system"
          },
          "rawValue": "31.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "os_world_external.csv:row=26;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 31.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "osworld",
          "benchmarkName": "OSWorld 2.0",
          "evidenceUrl": "https://os-world.github.io/",
          "line": 3989,
          "metricId": "Score",
          "observedAt": "2025-03-11",
          "protocol": {
            "harness": "computer-use-preview (100 steps)",
            "subject_type": "system"
          },
          "rawValue": "30.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "os_world_external.csv:row=28;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 30.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "osworld",
          "benchmarkName": "OSWorld 2.0",
          "evidenceUrl": "https://os-world.github.io/",
          "line": 3991,
          "metricId": "Score",
          "observedAt": "2025-03-11",
          "protocol": {
            "harness": "computer-use-preview (15 steps)",
            "subject_type": "system"
          },
          "rawValue": "26.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "os_world_external.csv:row=38;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 26.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "computer-use-preview-2025-03-11",
      "numericRowCount": 3,
      "observedAtMax": "2025-03-11",
      "observedAtMin": "2025-03-11",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1052,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.656",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=82;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.656
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2963,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.016",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=56;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3156,
          "metricId": "Overall accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.682",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=18;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "curie",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1048,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.722",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=64;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.722
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2958,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.09",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=30;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3157,
          "metricId": "Overall accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.775",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=19;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "davinci",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-bbh_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bbh_external",
          "benchmarkName": "BIG-Bench Hard (Epoch external aggregation)",
          "evidenceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf",
          "line": 983,
          "metricId": "Average",
          "observedAt": "2024-02-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bbh_external.csv:row=92;column=Average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.84
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3693,
          "metricId": "EM",
          "observedAt": "2024-02-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.827",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=65;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 82.69999999999999
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf",
          "line": 3695,
          "metricId": "EM",
          "observedAt": "2024-02-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.819",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=68;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 81.89999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Average",
        "EM"
      ],
      "modelRef": "gemini-1.5-pro-001-feb24",
      "numericRowCount": 3,
      "observedAtMax": "2024-02-15",
      "observedAtMin": "2024-02-15",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 661,
          "metricId": "Score",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.323",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=172;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.323
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2117,
          "metricId": "ECI Score",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=518;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.84
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3377,
          "metricId": "Mean score",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "7.65",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=13;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.65
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Mean score",
        "Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-04-17 (24K thinking)",
      "numericRowCount": 3,
      "observedAtMax": "2025-04-17",
      "observedAtMin": "2025-04-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 431,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0216",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=155;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0216
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 694,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.16",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=205;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.16
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2202,
          "metricId": "ECI Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=656;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20_1K",
      "numericRowCount": 3,
      "observedAtMax": "2025-05-20",
      "observedAtMin": "2025-05-20",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 432,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0212",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=156;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0212
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 675,
          "metricId": "Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2583",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=186;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.2583
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2197,
          "metricId": "ECI Score",
          "observedAt": "2025-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=650;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20_8K",
      "numericRowCount": 3,
      "observedAtMax": "2025-05-20",
      "observedAtMin": "2025-05-20",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 425,
          "metricId": "Score",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0292",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=149;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0292
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 666,
          "metricId": "Score",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.295",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=177;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.295
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2193,
          "metricId": "ECI Score",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "145.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=645;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 145.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "gemini-2.5-pro_8K",
      "numericRowCount": 3,
      "observedAtMax": "2025-06-17",
      "observedAtMin": "2025-06-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index",
        "epoch-mystery_game_puzzles"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1579,
          "metricId": "Average score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.498",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=12;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 49.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1824,
          "metricId": "ECI Score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "151.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=116;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 151.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mystery_game_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3876,
          "metricId": "mean_score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.26",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mystery_game_puzzles.csv:row=23;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 26.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Average score",
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemini-3-flash-preview_low",
      "numericRowCount": 3,
      "observedAtMax": "2025-12-17",
      "observedAtMin": "2025-12-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index",
        "epoch-mystery_game_puzzles"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1583,
          "metricId": "Average score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.49",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=16;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 49.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1823,
          "metricId": "ECI Score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "151.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=115;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 151.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mystery_game_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3874,
          "metricId": "mean_score",
          "observedAt": "2025-12-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.25",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mystery_game_puzzles.csv:row=21;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Average score",
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gemini-3-flash-preview_minimal",
      "numericRowCount": 3,
      "observedAtMax": "2025-12-17",
      "observedAtMin": "2025-12-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-vending_bench_2_external",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1896,
          "metricId": "ECI Score",
          "observedAt": "2026-02-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "154.61",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=189;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 154.61
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-vending_bench_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5215,
          "metricId": "Score",
          "observedAt": "2026-02-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "3774.254",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "vending_bench_2_external.csv:row=36;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 3774.254
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4825,
          "metricId": "mean_score",
          "observedAt": "2026-02-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.756198347107438",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "swe_bench_verified.csv:row=14;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 75.6198347107438
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score",
        "mean_score"
      ],
      "modelRef": "gemini-3.1-pro-preview-customtools",
      "numericRowCount": 3,
      "observedAtMax": "2026-02-19",
      "observedAtMin": "2026-02-19",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-mindcube_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1494,
          "metricId": "Accuracy",
          "observedAt": "2025-03-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=133;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mindcube_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3627,
          "metricId": "Overall score",
          "observedAt": "2025-03-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4667",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mindcube_external.csv:row=2;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.67
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4539,
          "metricId": "Score",
          "observedAt": "2025-03-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.173611111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=107;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.173611111111111
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Overall score",
        "Score"
      ],
      "modelRef": "gemma-3-12b-it",
      "numericRowCount": 3,
      "observedAtMax": "2025-03-12",
      "observedAtMin": "2025-03-12",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-lech_mazur_writing_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 185,
          "metricId": "Performance",
          "observedAt": "2025-08-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "344.82",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=99;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 344.82
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3404,
          "metricId": "Mean score",
          "observedAt": "2025-08-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "7.34",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=40;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.34
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3292,
          "metricId": "Accuracy",
          "observedAt": "2025-08-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.08320000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hle_external.csv:row=32;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Mean score",
        "Performance"
      ],
      "modelRef": "glm-4.5",
      "numericRowCount": 3,
      "observedAtMax": "2025-08-03",
      "observedAtMin": "2025-08-03",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-hella_swag_external",
        "epoch-wino_grande_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2068,
          "metricId": "ECI Score",
          "observedAt": "2023-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "126.19",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=422;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 126.19
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 3196,
          "metricId": "Overall accuracy",
          "observedAt": "2023-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.953",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=59;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 95.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-wino_grande_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2311.16867",
          "line": 5648,
          "metricId": "Accuracy",
          "observedAt": "2023-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.875",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "wino_grande_external.csv:row=94;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 87.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Overall accuracy"
      ],
      "modelRef": "gpt-4-32k-0314",
      "numericRowCount": 3,
      "observedAtMax": "2023-03-14",
      "observedAtMin": "2023-03-14",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-metr_time_horizons_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2217,
          "metricId": "ECI Score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "125.98",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=741;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 125.98
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3622,
          "metricId": "average_score",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.271786",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=47;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.271786
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3716,
          "metricId": "EM",
          "observedAt": "2023-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.796",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=91;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.60000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "average_score"
      ],
      "modelRef": "gpt-4-turbo",
      "numericRowCount": 3,
      "observedAtMax": "2023-11-06",
      "observedAtMin": "2023-11-06",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "frontiermath"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1889,
          "metricId": "ECI Score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "161.71",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=182;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 161.71
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2506,
          "metricId": "mean_score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.396",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=13;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "frontiermath",
          "benchmarkName": "FrontierMath T1–T3",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2404,
          "metricId": "mean_score",
          "observedAt": "2026-04-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.524",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath.csv:row=12;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.400000000000006
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "gpt-5.5-pro-pre-release_high",
      "numericRowCount": 3,
      "observedAtMax": "2026-04-23",
      "observedAtMin": "2026-04-23",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 16,
          "metricId": "Percent correct",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "41.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=2;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2098,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=487;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5463,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4817",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=70;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 48.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Percent correct"
      ],
      "modelRef": "gpt-oss-120b_high",
      "numericRowCount": 3,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-terminalbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2246,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=798;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5020,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-01",
          "protocol": {
            "harness": "Terminus 2",
            "subject_type": "system"
          },
          "rawValue": "0.187",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=176;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 18.7
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5033,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.142",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=189;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy mean",
        "ECI Score"
      ],
      "modelRef": "gpt-oss-120b_unknown",
      "numericRowCount": 3,
      "observedAtMax": "2025-11-03",
      "observedAtMin": "2025-08-05",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-terminalbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2247,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.81",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=799;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.81
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5047,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-03",
          "protocol": {
            "harness": "Mini-SWE-Agent",
            "subject_type": "system"
          },
          "rawValue": "0.034",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=203;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.4000000000000004
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-terminalbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5048,
          "metricId": "Accuracy mean",
          "observedAt": "2025-11-01",
          "protocol": {
            "harness": "Terminus 2",
            "subject_type": "system"
          },
          "rawValue": "0.031000000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "terminalbench_external.csv:row=204;column=Accuracy mean",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.1000000000000005
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy mean",
        "ECI Score"
      ],
      "modelRef": "gpt-oss-20b_unknown",
      "numericRowCount": 3,
      "observedAtMax": "2025-11-03",
      "observedAtMin": "2025-08-05",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 473,
          "metricId": "Score",
          "observedAt": "2025-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0042",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=197;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0042
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 693,
          "metricId": "Score",
          "observedAt": "2025-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.165",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=204;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.165
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2201,
          "metricId": "ECI Score",
          "observedAt": "2025-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=655;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.04
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "grok-3-mini_low",
      "numericRowCount": 3,
      "observedAtMax": "2025-06-24",
      "observedAtMin": "2025-06-24",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index",
        "epoch-gdpval_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 83,
          "metricId": "Percent correct",
          "observedAt": "2025-07-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "79.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=74;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2112,
          "metricId": "ECI Score",
          "observedAt": "2025-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "146.91",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=511;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 146.91
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gdpval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2642,
          "metricId": "Win Rate (%)",
          "observedAt": "2025-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.21100000000000002",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gdpval_external.csv:row=11;column=Win Rate (%)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Percent correct",
        "Win Rate (%)"
      ],
      "modelRef": "grok-4-0709_high",
      "numericRowCount": 3,
      "observedAtMax": "2025-07-11",
      "observedAtMin": "2025-07-09",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-gbaeval_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1409,
          "metricId": "Accuracy",
          "observedAt": "2026-05-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0914285714285714",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=48;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.14285714285714
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gbaeval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2598,
          "metricId": "Overall score",
          "observedAt": "2026-05-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.023812535358844113",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gbaeval_external.csv:row=16;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.381253535884411
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4512,
          "metricId": "Score",
          "observedAt": "2026-05-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.502314814814815",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=80;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.502314814814815
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Overall score",
        "Score"
      ],
      "modelRef": "grok-build-0.1",
      "numericRowCount": 3,
      "observedAtMax": "2026-05-29",
      "observedAtMin": "2026-05-29",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 364,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3305555555555556",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=88;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.3305555555555556
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 583,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.78",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=92;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.78
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2174,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=625;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "inkling-small_high",
      "numericRowCount": 3,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 407,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04861111111111112",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=131;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.04861111111111112
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 627,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.525",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=137;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.525
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2179,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=630;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "inkling-small_low",
      "numericRowCount": 3,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 380,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.13611111111111113",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=104;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.13611111111111113
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 601,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.67",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=111;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.67
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2176,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=627;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "inkling-small_medium",
      "numericRowCount": 3,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 419,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04027777777777778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=143;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.04027777777777778
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 632,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.485",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=142;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.485
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2180,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=631;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "inkling-small_minimal",
      "numericRowCount": 3,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 493,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=217;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 717,
          "metricId": "Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.055",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=228;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.055
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2208,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.17",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=663;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "inkling-small_none",
      "numericRowCount": 3,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-proofbench_external",
        "epoch-surface_evolver_bench_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-proofbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4419,
          "metricId": "Accuracy",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "proofbench_external.csv:row=59;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-surface_evolver_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4809,
          "metricId": "Mean score",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.15625",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "surface_evolver_bench_external.csv:row=23;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 15.625
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-webdev_arena_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5368,
          "metricId": "Arena Score",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1347.31",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "webdev_arena_external.csv:row=93;column=Arena Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "elo",
          "value": 1347.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Arena Score",
        "Mean score"
      ],
      "modelRef": "laguna-m.1",
      "numericRowCount": 3,
      "observedAtMax": "2026-04-28",
      "observedAtMin": "2026-04-28",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1038,
          "metricId": "Score",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.85",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=35;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.85
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2138,
          "metricId": "ECI Score",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "100.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=550;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 100.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2948,
          "metricId": "EM",
          "observedAt": "2023-06-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.344",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=11;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 34.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "EM",
        "Score"
      ],
      "modelRef": "mpt-30b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2023-06-22",
      "observedAtMin": "2023-06-22",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 436,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.020500000000000004",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=160;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.020500000000000004
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 637,
          "metricId": "Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4433000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=147;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4433000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2182,
          "metricId": "ECI Score",
          "observedAt": "2025-06-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "147.51",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=633;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 147.51
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "o3-pro-2025-06-10_low",
      "numericRowCount": 3,
      "observedAtMax": "2025-06-10",
      "observedAtMin": "2025-06-10",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2009,
          "metricId": "ECI Score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.31",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=325;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.31
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3566,
          "metricId": "mean_score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.1082892749244713",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=99;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 10.82892749244713
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2933,
          "metricId": "mean_score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2989267676767677",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=259;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.892676767676768
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "open-mistral-nemo-2407",
      "numericRowCount": 3,
      "observedAtMax": "2024-07-18",
      "observedAtMin": "2024-07-18",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2007,
          "metricId": "ECI Score",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "121.22",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=321;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 121.22
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3568,
          "metricId": "mean_score",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.24244712990936557",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=101;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 24.244712990936556
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2928,
          "metricId": "mean_score",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.34059343434343436",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=254;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 34.05934343434344
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "open-mixtral-8x22b",
      "numericRowCount": 3,
      "observedAtMax": "2024-04-17",
      "observedAtMin": "2024-04-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2008,
          "metricId": "ECI Score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=323;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.04
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3567,
          "metricId": "mean_score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.09950906344410876",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=100;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 9.950906344410877
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2931,
          "metricId": "mean_score",
          "observedAt": "2023-12-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.29829545454545453",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=257;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.829545454545453
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "open-mixtral-8x7b",
      "numericRowCount": 3,
      "observedAtMax": "2023-12-11",
      "observedAtMin": "2023-12-11",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1501,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=140;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2148,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=592;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4562,
          "metricId": "Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.35995370370370405",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=130;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.35995370370370405
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "ECI Score",
        "Score"
      ],
      "modelRef": "openai/gpt-oss-120b_low",
      "numericRowCount": 3,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1042,
          "metricId": "Score",
          "observedAt": "2022-05-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.793",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=46;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.793
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2960,
          "metricId": "EM",
          "observedAt": "2022-05-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=41;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3229,
          "metricId": "Overall accuracy",
          "observedAt": "2022-05-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.791",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=101;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.10000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "opt-175b",
      "numericRowCount": 3,
      "observedAtMax": "2022-05-02",
      "observedAtMin": "2022-05-02",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1045,
          "metricId": "Score",
          "observedAt": "2022-05-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.76",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=55;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.76
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2962,
          "metricId": "EM",
          "observedAt": "2022-05-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.018",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=52;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.7999999999999998
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3228,
          "metricId": "Overall accuracy",
          "observedAt": "2022-05-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.745",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=100;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 74.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "opt-66b",
      "numericRowCount": 3,
      "observedAtMax": "2022-05-03",
      "observedAtMin": "2022-05-03",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3497,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6527567975830816",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=30;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 65.27567975830816
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4190,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.17777777777777778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=183;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 17.77777777777778
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2856,
          "metricId": "mean_score",
          "observedAt": "2025-01-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4810606060606061",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=182;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 48.10606060606061
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwen-plus-2025-01-25",
      "numericRowCount": 3,
      "observedAtMax": "2025-01-25",
      "observedAtMin": "2025-01-25",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3496,
          "metricId": "mean_score",
          "observedAt": "2024-11-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5623111782477341",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=29;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.23111782477341
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4189,
          "metricId": "mean_score",
          "observedAt": "2024-11-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.06111111111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=182;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.111111111111111
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2855,
          "metricId": "mean_score",
          "observedAt": "2024-11-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.41792929292929293",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=181;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.792929292929294
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwen-turbo-2024-11-01",
      "numericRowCount": 3,
      "observedAtMax": "2024-11-01",
      "observedAtMin": "2024-11-01",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-math_level_5",
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3523,
          "metricId": "mean_score",
          "observedAt": "2024-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5607061933534743",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=56;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.07061933534743
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4224,
          "metricId": "mean_score",
          "observedAt": "2024-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.07361111111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=217;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.361111111111112
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2882,
          "metricId": "mean_score",
          "observedAt": "2024-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.46085858585858586",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=208;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.08585858585859
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwen2.5-32b-instruct",
      "numericRowCount": 3,
      "observedAtMax": "2024-09-17",
      "observedAtMin": "2024-09-17",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-lech_mazur_writing_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2118,
          "metricId": "ECI Score",
          "observedAt": "2025-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "133.39",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=520;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 133.39
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3385,
          "metricId": "Mean score",
          "observedAt": "2025-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "7.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=21;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.29
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3429,
          "metricId": "Global average",
          "observedAt": "2025-01-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "62.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=17;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 62.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "Global average",
        "Mean score"
      ],
      "modelRef": "qwen2.5-max",
      "numericRowCount": 3,
      "observedAtMax": "2025-01-28",
      "observedAtMin": "2025-01-28",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 158,
          "metricId": "Performance",
          "observedAt": "2026-05-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "694.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=72;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 694.12
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1435,
          "metricId": "Accuracy",
          "observedAt": "2026-05-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.022857142857142902",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=74;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.28571428571429
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4520,
          "metricId": "Score",
          "observedAt": "2026-05-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4004629629629631",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=88;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4004629629629631
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "Accuracy",
        "Performance",
        "Score"
      ],
      "modelRef": "step-3.7-flash",
      "numericRowCount": 3,
      "observedAtMax": "2026-05-29",
      "observedAtMin": "2026-05-29",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1056,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.464",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=89;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.464
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2967,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.004",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=67;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3253,
          "metricId": "Overall accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.429",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=125;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "text-ada-001",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1057,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.451",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=91;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.451
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2968,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=68;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3254,
          "metricId": "Overall accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.561",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=126;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.10000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "text-babbage-001",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external",
        "epoch-hella_swag_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1053,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.62",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=86;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.62
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2966,
          "metricId": "EM",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.006",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=65;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-hella_swag_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/hellaswag",
          "line": 3255,
          "metricId": "Overall accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.676",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hella_swag_external.csv:row=127;column=Overall accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 67.60000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "EM",
        "Overall accuracy",
        "Score"
      ],
      "modelRef": "text-curie-001",
      "numericRowCount": 3,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-frontiermath_tier_4",
        "frontiermath"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1929,
          "metricId": "ECI Score",
          "observedAt": "2025-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=222;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.29
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2532,
          "metricId": "mean_score",
          "observedAt": "2025-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.02127659574468085",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=39;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.127659574468085
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "frontiermath",
          "benchmarkName": "FrontierMath T1–T3",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2429,
          "metricId": "mean_score",
          "observedAt": "2025-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.03819444444444445",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath.csv:row=37;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.8194444444444446
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "zai-org/GLM-4.6",
      "numericRowCount": 3,
      "observedAtMax": "2025-09-30",
      "observedAtMin": "2025-09-30",
      "rowCount": 3,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 344,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Lite",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 3.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[79]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 3.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 80,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Test",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 1.96,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[19]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 1.96
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 260,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Verified",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 4.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[175]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 4.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 2",
      "numericRowCount": 3,
      "observedAtMax": "2023-10-10",
      "observedAtMin": "2023-10-10",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 68,
          "metricId": "resolved",
          "observedAt": "2024-11-21",
          "protocol": {
            "checked": false,
            "harness": "AutoCodeRover-v2.0",
            "leaderboard_variant": "Test",
            "scaffold": "AutoCodeRover-v2.0",
            "subject_type": "system"
          },
          "rawValue": 24.89,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[7]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 24.89
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 190,
          "metricId": "resolved",
          "observedAt": "2025-01-22",
          "protocol": {
            "checked": false,
            "harness": "AutoCodeRover-v2.1",
            "leaderboard_variant": "Verified",
            "scaffold": "AutoCodeRover-v2.1",
            "subject_type": "system"
          },
          "rawValue": 51.6,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[105]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 51.6
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 202,
          "metricId": "resolved",
          "observedAt": "2024-11-08",
          "protocol": {
            "checked": false,
            "harness": "AutoCodeRover-v2.0",
            "leaderboard_variant": "Verified",
            "scaffold": "AutoCodeRover-v2.0",
            "subject_type": "system"
          },
          "rawValue": 46.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[117]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 46.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 3.5-Sonnet-20241022",
      "numericRowCount": 3,
      "observedAtMax": "2025-01-22",
      "observedAtMin": "2024-11-08",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-multimodal",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 352,
          "metricId": "resolved",
          "observedAt": "2025-05-28",
          "protocol": {
            "checked": true,
            "harness": "OpenHands-Versa",
            "leaderboard_variant": "Multimodal",
            "scaffold": "OpenHands-Versa",
            "subject_type": "system"
          },
          "rawValue": 34.43,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[3]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 34.43
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 98,
          "metricId": "resolved",
          "observedAt": "2025-07-31",
          "protocol": {
            "checked": false,
            "harness": "Harness AI",
            "leaderboard_variant": "Verified",
            "scaffold": "Harness AI",
            "subject_type": "system"
          },
          "rawValue": 74.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[13]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 74.8
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 170,
          "metricId": "resolved",
          "observedAt": "2025-09-24",
          "protocol": {
            "checked": false,
            "harness": "Artemis Agent v2",
            "leaderboard_variant": "Verified",
            "scaffold": "Artemis Agent v2",
            "subject_type": "system"
          },
          "rawValue": 57.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[85]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 57.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude Sonnet 4",
      "numericRowCount": 3,
      "observedAtMax": "2025-09-24",
      "observedAtMin": "2025-05-28",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 287,
          "metricId": "resolved",
          "observedAt": "2024-10-25",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Lite",
            "scaffold": "OpenHands",
            "subject_type": "system"
          },
          "rawValue": 41.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[22]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 41.67
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 67,
          "metricId": "resolved",
          "observedAt": "2024-11-03",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Test",
            "scaffold": "OpenHands",
            "subject_type": "system"
          },
          "rawValue": 29.38,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[6]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 29.38
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 185,
          "metricId": "resolved",
          "observedAt": "2024-10-29",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Verified",
            "scaffold": "OpenHands",
            "subject_type": "system"
          },
          "rawValue": 53.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[100]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 53.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "CodeAct v2.1 (claude-3-5-sonnet-20241022)",
      "numericRowCount": 3,
      "observedAtMax": "2024-11-03",
      "observedAtMin": "2024-10-25",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 348,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Lite",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 0.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[83]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 0.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 84,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Test",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 0.17,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[23]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 0.17
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 264,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Verified",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 0.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[179]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 0.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-3.5",
      "numericRowCount": 3,
      "observedAtMax": "2023-10-10",
      "observedAtMin": "2023-10-10",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-multilingual",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 2,
          "metricId": "resolved",
          "observedAt": "2026-02-17",
          "protocol": {
            "checked": null,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 75.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[1]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 75.8
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multilingual",
          "benchmarkName": "SWE-bench Multilingual",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 48,
          "metricId": "resolved",
          "observedAt": "2026-02-13",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Multilingual",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 72.7,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[1]=Multilingual;results[0]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 72.7
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 93,
          "metricId": "resolved",
          "observedAt": "2026-02-17",
          "protocol": {
            "checked": null,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 75.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[8]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 75.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Gemini 3 Flash",
      "numericRowCount": 3,
      "observedAtMax": "2026-02-17",
      "observedAtMin": "2026-02-13",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-multilingual",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 15,
          "metricId": "resolved",
          "observedAt": "2026-02-26",
          "protocol": {
            "checked": null,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 69.6,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[14]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 69.6
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multilingual",
          "benchmarkName": "SWE-bench Multilingual",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 52,
          "metricId": "resolved",
          "observedAt": "2026-02-13",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Multilingual",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 68.7,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[1]=Multilingual;results[4]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 68.7
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 132,
          "metricId": "resolved",
          "observedAt": "2026-02-26",
          "protocol": {
            "checked": null,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 69.6,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[47]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 69.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Gemini 3 Pro",
      "numericRowCount": 3,
      "observedAtMax": "2026-02-26",
      "observedAtMin": "2026-02-13",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 6,
          "metricId": "resolved",
          "observedAt": "2025-11-18",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 74.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[5]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 74.2
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 105,
          "metricId": "resolved",
          "observedAt": "2025-11-18",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 74.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[20]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 74.2
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 88,
          "metricId": "resolved",
          "observedAt": "2025-11-20",
          "protocol": {
            "checked": null,
            "harness": "live-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "live-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 77.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[3]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 77.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Gemini 3 Pro Preview",
      "numericRowCount": 3,
      "observedAtMax": "2025-11-20",
      "observedAtMin": "2025-11-18",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 37,
          "metricId": "resolved",
          "observedAt": "2025-08-07",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 43.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[36]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 43.8
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 209,
          "metricId": "resolved",
          "observedAt": "2025-08-07",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 43.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[124]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 43.8
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 183,
          "metricId": "resolved",
          "observedAt": "2025-08-04",
          "protocol": {
            "checked": false,
            "harness": "CodeSweep - SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "CodeSweep - SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 53.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[98]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 53.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Kimi K2 Instruct",
      "numericRowCount": 3,
      "observedAtMax": "2025-08-07",
      "observedAtMin": "2025-08-04",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 347,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Lite",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 1.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[82]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 1.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 82,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Test",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 0.7,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[21]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 0.7
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 263,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Verified",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 1.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[178]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 1.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "SWE-Llama 13B",
      "numericRowCount": 3,
      "observedAtMax": "2023-10-10",
      "observedAtMin": "2023-10-10",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 3,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-test",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 346,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Lite",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 1.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[81]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 1.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-test",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 83,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Test",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 0.7,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[2]=Test;results[22]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 0.7
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 262,
          "metricId": "resolved",
          "observedAt": "2023-10-10",
          "protocol": {
            "checked": true,
            "harness": "RAG baseline",
            "leaderboard_variant": "Verified",
            "scaffold": "RAG baseline",
            "subject_type": "system"
          },
          "rawValue": 1.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[177]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 1.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 3
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "SWE-Llama 7B",
      "numericRowCount": 3,
      "observedAtMax": "2023-10-10",
      "observedAtMin": "2023-10-10",
      "rowCount": 3,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 3
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/openpangu/openPangu-2.0-Flash",
          "line": 15,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/openPangu-2.0-Flash.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 93.3,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=15;row=14",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 93.3
        },
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/openpangu/openPangu-2.0-Flash",
          "line": 22,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/openPangu-2.0-Flash.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.5,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=22;row=21",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 86.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openpangu/openPangu-2.0-Flash",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 55,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=55;row=54",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 82.7
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 69,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.23,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=69;row=68",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "RedHatAI/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 58,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=58;row=57",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 82.7
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 72,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.23,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=72;row=71",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-MXFP4_MOE-dequant-bf16-vllm",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 56,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=56;row=55",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 82.7
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 70,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.23,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=70;row=69",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q3_K_M-dequant-bf16-vllm",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 57,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=57;row=56",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 82.7
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 71,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.23,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=71;row=70",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q4_K_M-dequant-bf16-vllm",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 54,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=54;row=53",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 82.7
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 68,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.23,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=68;row=67",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/openpangu/openPangu-2.0-Flash",
          "line": 49,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/openPangu-2.0-Flash.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=49;row=48",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 83.7
        },
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/openpangu/openPangu-2.0-Flash",
          "line": 67,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/openPangu-2.0-Flash.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.8,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=67;row=66",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openpangu/openPangu-2.0-Flash",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 30,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 34.7,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=30;row=29",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 34.7
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 6,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 54,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=6;row=5",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 54.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 53,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.82,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=53;row=52",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.82
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 67,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.26,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=67;row=66",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 18.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "RedHatAI/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 56,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.82,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=56;row=55",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.82
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 70,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.26,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=70;row=69",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 18.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-MXFP4_MOE-dequant-bf16-vllm",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 54,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.82,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=54;row=53",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.82
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 68,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.26,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=68;row=67",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 18.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q3_K_M-dequant-bf16-vllm",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 55,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.82,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=55;row=54",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.82
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 69,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.26,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=69;row=68",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 18.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q4_K_M-dequant-bf16-vllm",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2-Thinking",
          "line": 19,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 44.9,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=19;row=18",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 44.9
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2-Thinking",
          "line": 50,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 23.9,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=50;row=49",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 23.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Thinking",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 52,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.82,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=52;row=51",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.82
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 66,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.26,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=66;row=65",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 18.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
          "line": 25,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.4,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=25;row=24",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 37.4
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
          "line": 42,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 26.7,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=42;row=41",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 26.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
          "line": 26,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.4,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=26;row=25",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 37.4
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
          "line": 43,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 26.1,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=43;row=42",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 26.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-35B-A3B",
          "line": 32,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-35b-a3b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 33.4,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=32;row=31",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 33.4
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-35B-A3B",
          "line": 44,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-35b-a3b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 25.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=44;row=43",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 25.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-35B-A3B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-397B",
          "line": 1,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-397b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 56.1,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=1;row=0",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 56.1
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-397B",
          "line": 20,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-397b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 44.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=20;row=19",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 44.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-397B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-9B",
          "line": 37,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-9b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 30.5,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=37;row=36",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 30.5
        },
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-9B",
          "line": 63,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-9b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 20.2,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=63;row=62",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 20.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-9B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/openpangu/openPangu-2.0-Flash",
          "line": 50,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/openPangu-2.0-Flash.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 63.1,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=50;row=49",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 63.1
        },
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/openpangu/openPangu-2.0-Flash",
          "line": 65,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/openPangu-2.0-Flash.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 57.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=65;row=64",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 57.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openpangu/openPangu-2.0-Flash",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 386,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 948.5118121152566,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=385",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 948.5118121152566
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 752,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 909.5914901400286,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=751",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 909.5914901400286
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "RWKV-4-Raven-14B",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 388,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 932.9364104341644,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=387",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 932.9364104341644
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 757,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 780.9277882162605,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=756",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 780.9277882162605
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "alpaca-13b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 173,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1364.469228199651,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=172",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1364.469228199651
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 944,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1367.618517955787,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=943",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1367.618517955787
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-experimental-chat-10-09",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 138,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1398.1901884662207,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=137",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1398.1901884662207
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 846,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1446.8533379359253,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=845",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1446.8533379359253
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-experimental-chat-26-01-10",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 293,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1208.5294177775695,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=292",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1208.5294177775695
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 670,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1208.6265952892707,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=669",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1208.6265952892707
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "amazon-nova-micro-v1.0",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 263,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1265.2716784131171,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=262",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1265.2716784131171
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 649,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1236.5527339386886,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=648",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1236.5527339386886
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "athene-70b-0725",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 287,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1224.1327822357496,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=286",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1224.1327822357496
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 669,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1209.5916623411936,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=668",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1209.5916623411936
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "c4ai-aya-expanse-32b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 307,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1185.2332427802814,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=306",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1185.2332427802814
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 684,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1179.9860240273845,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=683",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1179.9860240273845
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "c4ai-aya-expanse-8b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1131,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 995.8835629800434,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=130",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 995.8835629800434
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1647,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 999.4214056749179,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=646",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 999.4214056749179
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "c4ai-aya-vision-32b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 389,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 918.5902395594201,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=388",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 918.5902395594201
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 712,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1076.33501461236,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=711",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1076.33501461236
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "chatglm-6b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 383,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 972.1647297334553,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=382",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 972.1647297334553
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 720,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1063.16128960829,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=719",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1063.16128960829
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "chatglm3-6b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 362,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1065.6116808549991,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=361",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1065.6116808549991
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 746,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 975.4659517936313,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=745",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 975.4659517936313
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "codellama-34b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1143,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 927.7609587633187,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=142",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 927.7609587633187
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1653,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 970.0251798985842,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=652",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 970.0251798985842
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "cogvlm2-llama3-chat-19b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 316,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1163.265488483717,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=315",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1163.265488483717
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 692,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1145.4497482459,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=691",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1145.4497482459
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "command-r",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 303,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1187.4887990080958,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=302",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1187.4887990080958
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 683,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1181.237759402,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=682",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1181.237759402
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "command-r-08-2024",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 296,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1203.9622381465135,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=295",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1203.9622381465135
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 678,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1189.9868637267148,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=677",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1189.9868637267148
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "command-r-plus",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 280,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1229.028409741369,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=279",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1229.028409741369
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 655,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1225.4217913461034,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=654",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1225.4217913461034
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "command-r-plus-08-2024",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 335,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1118.998118009074,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=334",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1118.998118009074
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 718,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1066.831272029148,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=717",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1066.831272029148
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "dbrx-instruct-preview",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 301,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1191.323347028812,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=300",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1191.323347028812
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 674,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1199.8663069372737,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=673",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1199.8663069372737
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-coder-v2",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 343,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1105.1583415569248,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=342",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1105.1583415569248
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 701,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1130.8214132809117,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=700",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1130.8214132809117
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "deepseek-llm-67b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 393,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 851.2531590056515,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=392",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 851.2531590056515
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 755,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 835.412345922608,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=754",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 835.412345922608
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "dolly-v2-12b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 391,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 894.5247224686792,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=390",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 894.5247224686792
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 758,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 717.8883523266875,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=757",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 717.8883523266875
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "fastchat-t5-3b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 252,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1278.6901119582164,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=251",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1278.6901119582164
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 623,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1272.9929921488333,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=622",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1272.9929921488333
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-advanced-0514",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 328,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1130.6677444138677,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=327",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1130.6677444138677
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 709,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1095.4192817794537,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=708",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1095.4192817794537
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-pro",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 322,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1148.9778604607934,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=321",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1148.9778604607934
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 702,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1123.060523260226,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=701",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1123.060523260226
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-pro-dev-api",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 379,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1022.3514192714215,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=378",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1022.3514192714215
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 740,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1010.5456425086883,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=739",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1010.5456425086883
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-1.1-2b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 349,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1094.1013848724479,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=348",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1094.1013848724479
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 723,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1059.5798919831896,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=722",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1059.5798919831896
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-1.1-7b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 278,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1231.499057906653,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=277",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1231.499057906653
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 661,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1220.012804487858,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=660",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1220.012804487858
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-2-27b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 320,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1155.8953686984855,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=319",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1155.8953686984855
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 700,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1131.1773024787708,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=699",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1131.1773024787708
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-2-2b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 294,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1207.51331137132,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=293",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1207.51331137132
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 681,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1183.988857619776,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=680",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1183.988857619776
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-2-9b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 283,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1227.1955728147127,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=282",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1227.1955728147127
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 658,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1224.1238756173293,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=657",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1224.1238756173293
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-2-9b-it-simpo",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 380,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1002.2742634937238,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=379",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1002.2742634937238
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 744,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 984.9060635366909,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=743",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 984.9060635366909
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-2b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 223,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1306.2419151320485,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=222",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1306.2419151320485
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 608,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1308.433199819657,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=607",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1308.433199819657
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-3n-e4b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 366,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1056.282605170445,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=365",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1056.282605170445
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 729,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1034.4645734564965,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=728",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1034.4645734564965
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-7b-it",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 284,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1226.0389988487862,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=283",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1226.0389988487862
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 654,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1227.538517541424,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=653",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1227.538517541424
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4-0520",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 206,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1330.9287448530495,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=205",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1330.9287448530495
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 558,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1388.202735734629,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=557",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1388.202735734629
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "glm-4-plus-0111",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 332,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1125.0506565447054,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=331",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1125.0506565447054
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 715,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1074.3343112857865,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=714",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1074.3343112857865
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-3.5-turbo-0125",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 348,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1094.249688587442,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=347",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1094.249688587442
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 739,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1011.1439756450603,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=738",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1011.1439756450603
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-3.5-turbo-1106",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 266,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1262.3330827378143,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=265",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1262.3330827378143
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 650,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1236.5346324776474,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=649",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1236.5346324776474
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4-0125-preview",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 295,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1206.1388093885482,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=294",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1206.1388093885482
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 682,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1182.9413213546195,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=681",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1182.9413213546195
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4-0314",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 306,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1186.1369531193984,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=305",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1186.1369531193984
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 697,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1135.164821150563,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=696",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1135.164821150563
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4-0613",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 264,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1263.566821408624,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=263",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1263.566821408624
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 641,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1241.4319902111843,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=640",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1241.4319902111843
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt-4-1106-preview",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 357,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1079.9710863627174,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=356",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1079.9710863627174
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 716,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1069.0283278605043,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=715",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1069.0283278605043
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "granite-3.0-2b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 347,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1096.5677096054133,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=346",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1096.5677096054133
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 721,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1062.1293838098118,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=720",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1062.1293838098118
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "granite-3.0-8b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 331,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1127.531647195669,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=330",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1127.531647195669
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 696,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1137.3622016808713,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=695",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1137.3622016808713
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "granite-3.1-2b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 321,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1149.649740322826,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=320",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1149.649740322826
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 693,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1143.890994864865,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=692",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1143.890994864865
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "granite-3.1-8b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 224,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1304.7024869868624,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=223",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1304.7024869868624
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 617,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1288.3565029028111,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=616",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1288.3565029028111
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-2-2024-08-13",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 250,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1280.8200188000392,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=249",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1280.8200188000392
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 631,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1261.4419832286203,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=630",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1261.4419832286203
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-2-mini-2024-08-13",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 255,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1274.0902601941239,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=254",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1274.0902601941239
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 604,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1318.7579987643492,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=603",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1318.7579987643492
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-standard-2025-02-10",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 298,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1202.0517594219427,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=297",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1202.0517594219427
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 644,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1240.5744210279333,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=643",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1240.5744210279333
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-standard-256k",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1130,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 999.3515441720104,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=129",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 999.3515441720104
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1640,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1021.0251665143167,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=639",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1021.0251665143167
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "hunyuan-standard-vision-2024-12-31",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 274,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1241.4852685808103,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=273",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1241.4852685808103
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 637,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1247.1450711159546,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=636",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1247.1450711159546
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ibm-granite-h-small",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 319,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1158.5738039644832,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=318",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1158.5738039644832
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 675,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1198.474263480215,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=674",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1198.474263480215
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "internlm2_5-20b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 276,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1237.155625514207,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=275",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1237.155625514207
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 673,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1202.0439918771508,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=672",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1202.0439918771508
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "jamba-1.5-large",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 304,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1186.8071396263233,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=303",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1186.8071396263233
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 698,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1134.610782277584,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=697",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1134.610782277584
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "jamba-1.5-mini",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 382,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 989.7351294360024,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=381",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 989.7351294360024
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 754,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 879.9289492881965,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=753",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 879.9289492881965
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "koala-13b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 352,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1084.5856401874312,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=351",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1084.5856401874312
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 742,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1000.0304690932069,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=741",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1000.0304690932069
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-2-13b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 336,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1115.3569793323081,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=335",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1115.3569793323081
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 743,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 993.6518623490988,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=742",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 993.6518623490988
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-2-70b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 369,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1053.5261758626,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=368",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1053.5261758626
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 747,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 971.9750632375112,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=746",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 971.9750632375112
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-2-7b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 288,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1220.9281038236666,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=287",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1220.9281038236666
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 705,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1113.2409877653622,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=704",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1113.2409877653622
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3-70b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 313,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1165.7343614524696,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=312",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1165.7343614524696
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 714,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1075.0670411854867,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=713",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1075.0670411854867
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3-8b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 248,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1281.9308979487716,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=247",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1281.9308979487716
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 651,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1233.4218936967914,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=650",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1233.4218936967914
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-405b-instruct-fp8",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 268,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1261.0364800664431,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=267",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1261.0364800664431
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 665,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1214.2598745772848,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=664",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1214.2598745772848
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-70b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 305,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1186.7055209327045,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=304",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1186.7055209327045
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 690,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1149.7913845041098,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=689",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1149.7913845041098
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-8b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 282,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1228.1156063793965,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=281",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1228.1156063793965
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 685,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1178.6071147069388,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=684",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1178.6071147069388
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-nemotron-51b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 244,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1282.778464389829,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=243",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1282.778464389829
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 630,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1262.025982838198,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=629",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1262.025982838198
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-nemotron-70b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 214,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1319.5047050296778,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=213",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1319.5047050296778
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 985,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1312.4088283744982,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=984",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1312.4088283744982
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-nemotron-ultra-253b-v1",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 270,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1255.8489757954158,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=269",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1255.8489757954158
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 636,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1248.2587081819192,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=635",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1248.2587081819192
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-tulu-3-70b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 300,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1193.4312799456818,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=299",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1193.4312799456818
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 687,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1174.6441474915264,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=686",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1174.6441474915264
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.1-tulu-3-8b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 367,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1054.7628550735276,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=366",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1054.7628550735276
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 749,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 957.6763452583194,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=748",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 957.6763452583194
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.2-1b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 339,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1109.6625016531702,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=338",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1109.6625016531702
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 734,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1016.2867084337751,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=733",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1016.2867084337751
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.2-3b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 254,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1274.9015506043663,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=253",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1274.9015506043663
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 664,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1215.0820491095542,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=663",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1215.0820491095542
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-3.3-70b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1144,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 923.0444208230434,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=143",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 923.0444208230434
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1654,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 962.8920533557814,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=653",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 962.8920533557814
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llava-onevision-qwen2-72b-ov",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 247,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1281.94418094451,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=246",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1281.94418094451
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 973,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1326.0637516357115,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=972",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1326.0637516357115
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mercury",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1145,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 911.8394573905741,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=144",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 911.8394573905741
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1657,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 949.7953327958553,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=656",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 949.7953327958553
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "minicpm-v-2_6",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 302,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1191.040978565215,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=301",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1191.040978565215
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 676,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1192.402629555668,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=675",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1192.402629555668
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ministral-8b-2410",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 378,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1023.6348563384142,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=377",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1023.6348563384142
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 751,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 930.4826574687138,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=750",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 930.4826574687138
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-7b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 351,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1089.6936074928153,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=350",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1089.6936074928153
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 741,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1008.3358204444637,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=740",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1008.3358204444637
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-7b-instruct-v0.2",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 308,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1176.6083402798122,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=307",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1176.6083402798122
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 703,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1119.3136592963983,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=702",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1119.3136592963983
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-large-2402",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 261,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1266.272271310172,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=260",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1266.272271310172
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 646,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1239.3821098118513,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=645",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1239.3821098118513
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-large-2407",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 262,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1265.398030396689,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=261",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1265.398030396689
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 648,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1238.4541330308264,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=647",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1238.4541330308264
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-large-2411",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 314,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1165.2681789372982,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=313",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1165.2681789372982
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 707,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1108.0915165508386,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=706",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1108.0915165508386
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-medium",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 277,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1233.6267249952418,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=276",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1233.6267249952418
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 672,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1203.3882166428084,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=671",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1203.3882166428084
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mistral-small-24b-instruct-2501",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 317,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1162.1172252347938,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=316",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1162.1172252347938
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 704,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1115.3711175330425,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=703",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1115.3711175330425
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mixtral-8x22b-instruct-v0.1",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 327,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1131.688458795435,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=326",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1131.688458795435
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 724,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1054.3866118082383,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=723",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1054.3866118082383
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mixtral-8x7b-instruct-v0.1",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 385,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 955.3449638966019,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=384",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 955.3449638966019
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 750,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 953.7309947353674,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=749",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 953.7309947353674
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mpt-7b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 286,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1224.944288600695,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=285",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1224.944288600695
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 666,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1212.8066664003607,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=665",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1212.8066664003607
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nemotron-4-340b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 198,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1337.8259221395979,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=197",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1337.8259221395979
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 960,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1355.6845971818764,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=959",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1355.6845971818764
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nvidia-llama-3.3-nemotron-super-49b-v1.5",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 209,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1328.4011898031808,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=208",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1328.4011898031808
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 930,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "coding",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1379.924785784042,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=929",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1379.924785784042
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nvidia-nemotron-3.5-lightning-30b-a3b-nvfp4",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1141,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 940.3778582665976,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=140",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 940.3778582665976
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1655,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 960.1402560664769,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=654",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 960.1402560664769
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nvila-internal-15b-v1",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 390,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 916.1921623075388,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=389",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 916.1921623075388
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 756,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 801.9769991356608,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=755",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 801.9769991356608
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "oasst-pythia-12b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 291,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1217.9985237767903,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=290",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1217.9985237767903
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 677,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1190.853417773822,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=676",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1190.853417773822
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "olmo-2-0325-32b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 259,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1270.0416456197495,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=258",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1270.0416456197495
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 632,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1256.2726755610609,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=631",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1256.2726755610609
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "olmo-3.1-32b-think",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 375,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1031.9383417909596,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=374",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1031.9383417909596
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 737,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1012.608690961797,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=736",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1012.608690961797
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "olmo-7b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 346,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1097.1608726017346,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=345",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1097.1608726017346
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 717,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1069.015834035688,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=716",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1069.015834035688
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "openchat-3.5",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 341,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1106.7374371578271,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=340",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1106.7374371578271
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 710,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1082.7447690914587,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=709",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1082.7447690914587
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "openchat-3.5-0106",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 350,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1092.6278101406747,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=349",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1092.6278101406747
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 736,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1013.3317317844651,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=735",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1013.3317317844651
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "openhermes-2.5-mistral-7b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 377,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1027.3647777289086,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=376",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1027.3647777289086
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 753,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 886.2695488662062,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=752",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 886.2695488662062
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "palm-2",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 324,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1137.6804592573167,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=323",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1137.6804592573167
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 708,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1107.1269292457214,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=707",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1107.1269292457214
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3-medium-4k-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 372,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1050.5171903328737,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=371",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1050.5171903328737
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 735,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1015.3067895968651,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=734",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1015.3067895968651
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3-mini-128k-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 359,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1073.4345089114668,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=358",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1073.4345089114668
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 733,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1020.0283351106525,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=732",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1020.0283351106525
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3-mini-4k-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 356,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1080.085092644044,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=355",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1080.085092644044
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 730,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1032.4256245591819,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=729",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1032.4256245591819
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3-mini-4k-instruct-june-2024",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 338,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1109.9522542016257,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=337",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1109.9522542016257
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 722,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1059.7756125148478,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=721",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1059.7756125148478
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3-small-8k-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1147,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 811.9395305026102,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=146",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 811.9395305026102
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1659,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 838.2471137490459,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=658",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 838.2471137490459
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3-vision-128k-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1146,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 851.6866670530422,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=145",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 851.6866670530422
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1658,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 876.1577770449428,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=657",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 876.1577770449428
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-3.5-vision-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 292,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1216.790967281628,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=291",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1216.790967281628
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 668,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1211.0922629021259,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=667",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1211.0922629021259
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "phi-4",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 371,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1051.1548903664004,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=370",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1051.1548903664004
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 713,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1075.9081810020011,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=712",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1075.9081810020011
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen-14b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 246,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1282.0068890828863,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=245",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1282.0068890828863
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 634,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1253.0162378321706,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=633",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1253.0162378321706
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen-max-0919",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1120,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1057.8051998797039,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=119",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1057.8051998797039
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1631,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1073.1534719207255,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=630",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1073.1534719207255
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen-vl-max-1119",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 309,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1174.7010840238001,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=308",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1174.7010840238001
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 671,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1204.7945004267176,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=670",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1204.7945004267176
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen1.5-110b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 330,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1127.8231190214888,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=329",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1127.8231190214888
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 691,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1146.0755487574584,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=690",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1146.0755487574584
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen1.5-14b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 325,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1137.3322946105152,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=324",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1137.3322946105152
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 686,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1175.52705912306,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=685",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1175.52705912306
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen1.5-32b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 381,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 997.361686580062,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=380",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 997.361686580062
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 731,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1022.8809366213541,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=730",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1022.8809366213541
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen1.5-4b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 312,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1166.2822594388772,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=311",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1166.2822594388772
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 680,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1185.1813428074793,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=679",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1185.1813428074793
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen1.5-72b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 353,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1083.5883792311351,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=352",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1083.5883792311351
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 695,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1139.5544585225143,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=694",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1139.5544585225143
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen1.5-7b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 297,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1203.403366822571,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=296",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1203.403366822571
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 647,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1238.9225528697698,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=646",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1238.9225528697698
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2-72b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 279,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1230.1709262456743,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=278",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1230.1709262456743
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 660,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1220.8565673453286,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=659",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1220.8565673453286
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2.5-coder-32b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1612,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1159.2701445973807,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=611",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1159.2701445973807
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1099,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1152.5609076281462,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=98",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1152.5609076281462
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwen2.5-vl-32b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 318,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1161.9238487965247,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=317",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1161.9238487965247
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 663,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1216.8909231928756,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=662",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1216.8909231928756
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "qwq-32b-preview",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 273,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1248.510398604697,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=272",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1248.510398604697
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 645,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1240.064340181502,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=644",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1240.064340181502
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "reka-core-20240904",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 290,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1218.0495576887747,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=289",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1218.0495576887747
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 662,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1219.753286615543,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=661",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1219.753286615543
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "reka-flash-20240904",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 315,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1165.1183981337676,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=314",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1165.1183981337676
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 699,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1134.3464645913184,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=698",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1134.3464645913184
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "reka-flash-21b-20240226",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 311,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1170.56274663525,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=310",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1170.56274663525
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 694,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1142.218811882614,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=693",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1142.218811882614
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "reka-flash-21b-20240226-online",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 344,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1101.1044234357403,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=343",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1101.1044234357403
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 719,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1066.0944160144481,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=718",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1066.0944160144481
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "snowflake-arctic-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 340,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1107.7899084958235,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=339",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1107.7899084958235
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 726,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1044.2319576000539,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=725",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1044.2319576000539
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "starling-lm-7b-alpha",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 326,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1132.1727121174033,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=325",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1132.1727121174033
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 706,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1112.9748433226016,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=705",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1112.9748433226016
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "starling-lm-7b-beta",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1107,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1122.0492793777012,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=106",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1122.0492793777012
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1617,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1148.038948082251,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=616",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1148.038948082251
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "step-1o-vision-32k-highres",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1122,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1046.1824381869567,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=121",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1046.1824381869567
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1633,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1072.6594001982107,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=632",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1072.6594001982107
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "step-1v-32k",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 333,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1121.164145015469,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=332",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1121.164145015469
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 738,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1012.4962103482546,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=737",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1012.4962103482546
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "tulu-2-dpo-70b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 364,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1057.7856284893433,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=363",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1057.7856284893433
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 727,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1038.5321642669755,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=726",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1038.5321642669755
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "vicuna-13b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 342,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1105.3888628361722,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=341",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1105.3888628361722
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 728,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1034.8732258382627,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=727",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1034.8732258382627
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "vicuna-33b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 376,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1031.073789524972,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=375",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1031.073789524972
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 745,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 978.0436627735446,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=744",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 978.0436627735446
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "vicuna-7b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 358,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1076.6658676327984,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=357",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1076.6658676327984
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 732,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1021.7555453666297,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=731",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1021.7555453666297
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "wizardlm-13b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 334,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1119.649757062491,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=333",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1119.649757062491
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 725,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1051.4260011090705,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=724",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1051.4260011090705
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "wizardlm-70b",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 310,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1173.0883641408768,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=309",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1173.0883641408768
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 667,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1212.0296630036669,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=666",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1212.0296630036669
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "yi-1.5-34b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 329,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1128.5997056768867,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=328",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1128.5997056768867
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 688,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1174.6084276267711,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=687",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1174.6084276267711
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "yi-34b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-vision"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1136,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 977.9509880686868,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=135",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 977.9509880686868
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-vision",
          "benchmarkName": "Arena Vision",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 1644,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "vision",
            "category": "english",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1008.5246024447493,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=vision;split=latest;row_idx=643",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1008.5246024447493
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "yi-vision",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 360,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1069.549770348347,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=359",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1069.549770348347
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 748,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 958.9868836766037,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=747",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 958.9868836766037
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "zephyr-7b-beta",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 323,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1143.82218633614,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=322",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1143.82218633614
        },
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 711,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "chinese",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1081.004547602838,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=710",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1081.004547602838
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "zephyr-orpo-141b-A35b-v0.1",
      "numericRowCount": 2,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 2,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 9,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-22",
          "protocol": {
            "command": "aider --model openai/Qwen/Qwen2.5-Coder-32B-Instruct # via hyperbolic",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=209;dirname=2024-12-22-13-22-32--polyglot-qwen-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.4
        },
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 15,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-26",
          "protocol": {
            "command": "aider --model openai/Qwen2.5-Coder-32B-Instruct",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=365;dirname=2024-12-26-00-55-20--Qwen2.5-Coder-32B-Instruct",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Qwen2.5-Coder-32B-Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2024-12-26",
      "observedAtMin": "2024-12-22",
      "rowCount": 2,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 20,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "53.96%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=19;rank=20;model=GPT-4.1-2025-04-14 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 53.96
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 45,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "39.38%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=44;rank=45;model=GPT-4.1-2025-04-14 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 39.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "GPT-4.1-2025-04-14",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 27,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "50.45%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=26;rank=27;model=GPT-4.1-mini-2025-04-14 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 50.45
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 67,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "29.73%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=66;rank=67;model=GPT-4.1-mini-2025-04-14 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 29.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "GPT-4.1-mini-2025-04-14",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 58,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "33.05%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=57;rank=58;model=GPT-4.1-nano-2025-04-14 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 33.05
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 90,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "24.88%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=89;rank=90;model=GPT-4.1-nano-2025-04-14 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 24.88
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "GPT-4.1-nano-2025-04-14",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 52,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "36.87%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=51;rank=52;model=Gemini-2.5-Flash-Lite (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 36.87
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 73,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "28.03%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=72;rank=73;model=Gemini-2.5-Flash-Lite (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 28.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Gemini-2.5-Flash-Lite",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 3,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "72.51%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=2;rank=3;model=Gemini-3-Pro-Preview (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 72.51
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 7,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "68.14%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=6;rank=7;model=Gemini-3-Pro-Preview (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 68.14
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Gemini-3-Pro-Preview",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 9,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "62.97%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=8;rank=9;model=Grok-4-0709 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 62.97
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 10,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "61.38%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=9;rank=10;model=Grok-4-0709 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 61.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Grok-4-0709",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 48,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "37.69%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=47;rank=48;model=Mistral-Medium-2505;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 37.69
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 49,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "37.56%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=48;rank=49;model=Mistral-Medium-2505 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 37.56
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Mistral-Medium-2505",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 102,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "19.31%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=101;rank=102;model=Open-Mistral-Nemo-2407 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 19.31
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 78,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.63%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=77;rank=78;model=Open-Mistral-Nemo-2407 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.63
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Open-Mistral-Nemo-2407",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 92,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "23.93%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=91;rank=92;model=Qwen3-0.6B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 23.93
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 94,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "22.38%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=93;rank=94;model=Qwen3-0.6B (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 22.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-0.6B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 43,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "41.03%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=42;rank=43;model=Qwen3-14B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 41.03
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 47,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "37.77%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=46;rank=47;model=Qwen3-14B (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 37.77
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-14B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 23,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "52.15%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=22;rank=23;model=Qwen3-235B-A22B-Instruct-2507 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 52.15
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 31,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "47.99%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=30;rank=31;model=Qwen3-235B-A22B-Instruct-2507 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 47.99
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-235B-A22B-Instruct-2507",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 41,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "41.39%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=40;rank=41;model=Qwen3-30B-A3B-Instruct-2507 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 41.39
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 53,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "36.70%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=52;rank=53;model=Qwen3-30B-A3B-Instruct-2507 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 36.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-30B-A3B-Instruct-2507",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 29,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "48.71%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=28;rank=29;model=Qwen3-32B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 48.71
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 33,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "46.78%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=32;rank=33;model=Qwen3-32B (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 46.78
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-32B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 54,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "35.68%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=53;rank=54;model=Qwen3-4B-Instruct-2507 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 35.68
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 55,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "35.52%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=54;rank=55;model=Qwen3-4B-Instruct-2507 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 35.52
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-4B-Instruct-2507",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 39,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "42.57%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=38;rank=39;model=Qwen3-8B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 42.57
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 44,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "40.43%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=43;rank=44;model=Qwen3-8B (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 40.43
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-8B",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 46,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "38.37%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=45;rank=46;model=mistral-large-2411 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 38.37
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 63,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "31.84%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=62;rank=63;model=mistral-large-2411 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 31.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "mistral-large-2411",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 30,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "48.56%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=29;rank=30;model=o3-2025-04-16 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 48.56
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 8,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "63.05%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=7;rank=8;model=o3-2025-04-16 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 63.05
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "o3-2025-04-16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 21,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "53.24%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=20;rank=21;model=o4-mini-2025-04-16 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 53.24
        },
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 28,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "50.26%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=27;rank=28;model=o4-mini-2025-04-16 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 50.26
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "o4-mini-2025-04-16",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3505,
          "metricId": "mean_score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.8712235649546828",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=38;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 87.12235649546828
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2863,
          "metricId": "mean_score",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.44696969696969696",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=189;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.696969696969695
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "DeepSeek-R1-Distill-Qwen-14B",
      "numericRowCount": 2,
      "observedAtMax": "2025-01-20",
      "observedAtMin": "2025-01-20",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-balrog_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 883,
          "metricId": "Average progress",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.195",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=24;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 19.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3450,
          "metricId": "Global average",
          "observedAt": "2025-01-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "45.55",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=44;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.55
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average progress",
        "Global average"
      ],
      "modelRef": "DeepSeek-R1-Distill-Qwen-32B",
      "numericRowCount": 2,
      "observedAtMax": "2025-01-20",
      "observedAtMin": "2025-01-20",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2283,
          "metricId": "ECI Score",
          "observedAt": "2024-12-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "133.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=856;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 133.12
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-piqa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2412.19437",
          "line": 4293,
          "metricId": "Score",
          "observedAt": "2024-12-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.847",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "piqa_external.csv:row=70;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.847
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "DeepSeek-V3-Base",
      "numericRowCount": 2,
      "observedAtMax": "2024-12-26",
      "observedAtMin": "2024-12-26",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-simplebench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-simplebench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4636,
          "metricId": "Score (AVG@5)",
          "observedAt": "2025-12-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.526",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "simplebench_external.csv:row=50;column=Score (AVG@5)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.526
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5468,
          "metricId": "Accuracy",
          "observedAt": "2025-12-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4673",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=75;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score (AVG@5)"
      ],
      "modelRef": "DeepSeek-V3.2-Speciale",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-01",
      "observedAtMin": "2025-12-01",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2268,
          "metricId": "ECI Score",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "148.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=823;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 148.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5534,
          "metricId": "Accuracy",
          "observedAt": "2026-07-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3229",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=141;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "Inkling_high",
      "numericRowCount": 2,
      "observedAtMax": "2026-07-15",
      "observedAtMin": "2026-07-15",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-forecastbench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2319,
          "metricId": "Overall score",
          "observedAt": "2025-09-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=36;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5523,
          "metricId": "Accuracy",
          "observedAt": "2025-09-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3668",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=130;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 36.68
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Overall score"
      ],
      "modelRef": "Kimi-K2-Instruct-0905",
      "numericRowCount": 2,
      "observedAtMax": "2025-09-05",
      "observedAtMin": "2025-09-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 887,
          "metricId": "Average progress",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.168",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=28;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 16.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3753,
          "metricId": "EM",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.565",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=133;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.49999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average progress",
        "EM"
      ],
      "modelRef": "Llama-3.2-11B-Vision-Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2024-09-24",
      "observedAtMin": "2024-09-24",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-piqa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2284,
          "metricId": "ECI Score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "111.79",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=857;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 111.79
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-piqa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2401.04088",
          "line": 4320,
          "metricId": "Score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.822",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "piqa_external.csv:row=98;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.822
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "Mistral-7B-Instruct-v0.1",
      "numericRowCount": 2,
      "observedAtMax": "2023-09-27",
      "observedAtMin": "2023-09-27",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 885,
          "metricId": "Average progress",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.17600000000000002",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=26;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 17.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2232,
          "metricId": "ECI Score",
          "observedAt": "2024-07-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "118.31",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=770;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 118.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average progress",
        "ECI Score"
      ],
      "modelRef": "Mistral-Nemo-Instruct-2407",
      "numericRowCount": 2,
      "observedAtMax": "2024-07-18",
      "observedAtMin": "2024-07-18",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2270,
          "metricId": "ECI Score",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "121.22",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=827;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 121.22
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3789,
          "metricId": "EM",
          "observedAt": "2024-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=171;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "EM"
      ],
      "modelRef": "Mixtral-8x22B-v0.1",
      "numericRowCount": 2,
      "observedAtMax": "2024-04-17",
      "observedAtMin": "2024-04-17",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1490,
          "metricId": "Accuracy",
          "observedAt": "2025-02-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=129;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4540,
          "metricId": "Score",
          "observedAt": "2025-02-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.107638888888889",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=108;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.107638888888889
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "Phi-4-mini-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-27",
      "observedAtMin": "2025-02-27",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "aider-polyglot",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 62,
          "metricId": "Percent correct",
          "observedAt": "2025-03-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "20.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=51;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 20.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3419,
          "metricId": "Global average",
          "observedAt": "2025-03-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "71.96",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=6;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 71.96
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Global average",
        "Percent correct"
      ],
      "modelRef": "QwQ-32B",
      "numericRowCount": 2,
      "observedAtMax": "2025-03-06",
      "observedAtMin": "2025-03-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-forecastbench_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2340,
          "metricId": "Overall score",
          "observedAt": "2024-11-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "58.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=57;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 58.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3457,
          "metricId": "Global average",
          "observedAt": "2024-11-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "40.25",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=52;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.25
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Global average",
        "Overall score"
      ],
      "modelRef": "QwQ-32B-Preview",
      "numericRowCount": 2,
      "observedAtMax": "2024-11-28",
      "observedAtMin": "2024-11-28",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4575,
          "metricId": "Score",
          "observedAt": "2023-08-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.672",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=88;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.672
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/81a60d18e010b27b36cd465c6604b915-Paper-Conference.pdf",
          "line": 4585,
          "metricId": "Score",
          "observedAt": "2023-08-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.682",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=98;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.682
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "Qwen-VL-Chat",
      "numericRowCount": 2,
      "observedAtMax": "2023-08-20",
      "observedAtMin": "2023-08-20",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2221,
          "metricId": "ECI Score",
          "observedAt": "2024-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "112.73",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=752;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 112.73
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3066,
          "metricId": "EM",
          "observedAt": "2024-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.867",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=189;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 86.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "EM"
      ],
      "modelRef": "Qwen2.5-Coder-7B-Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2024-09-17",
      "observedAtMin": "2024-09-17",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-lech_mazur_writing_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3380,
          "metricId": "Mean score",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "7.53",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=16;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.53
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5537,
          "metricId": "Accuracy",
          "observedAt": "2025-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2975",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=144;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.75
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Mean score"
      ],
      "modelRef": "Qwen3-30B-A3B",
      "numericRowCount": 2,
      "observedAtMax": "2025-04-29",
      "observedAtMin": "2025-04-29",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 1139,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.764",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=203;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.764
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arxiv.org/pdf/1910.10683",
          "line": 4783,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.633",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=8;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.633
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "T5-Small",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3559,
          "metricId": "mean_score",
          "observedAt": "2024-04-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.25736404833836857",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=92;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.736404833836858
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2923,
          "metricId": "mean_score",
          "observedAt": "2024-04-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.43434343434343436",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=249;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.43434343434344
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "WizardLM-2-8x22B",
      "numericRowCount": 2,
      "observedAtMax": "2024-04-15",
      "observedAtMin": "2024-04-15",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3550,
          "metricId": "mean_score",
          "observedAt": "2024-05-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2548149546827795",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=83;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.48149546827795
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2909,
          "metricId": "mean_score",
          "observedAt": "2024-05-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.319760101010101",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=235;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 31.9760101010101
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "Yi-1.5-34B-Chat",
      "numericRowCount": 2,
      "observedAtMax": "2024-05-13",
      "observedAtMin": "2024-05-13",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-mmlu_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3639,
          "metricId": "EM",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.708",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=3;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 70.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3461,
          "metricId": "Global average",
          "observedAt": "2024-12-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "29.59",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=57;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.59
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "EM",
        "Global average"
      ],
      "modelRef": "amazon.nova-micro-v1:0",
      "numericRowCount": 2,
      "observedAtMax": "2024-12-03",
      "observedAtMin": "2024-12-03",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-mmlu_external",
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3666,
          "metricId": "EM",
          "observedAt": "2024-08-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.652",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=35;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 65.2
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3463,
          "metricId": "Global average",
          "observedAt": "2024-08-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "27.48",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=59;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.48
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "EM",
        "Global average"
      ],
      "modelRef": "c4ai-command-r-08-2024",
      "numericRowCount": 2,
      "observedAtMax": "2024-08-30",
      "observedAtMin": "2024-08-30",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1465,
          "metricId": "Accuracy",
          "observedAt": "2025-04-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=104;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2274,
          "metricId": "ECI Score",
          "observedAt": "2025-04-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "133.03",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=837;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 133.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "chutes/Llama-4-Maverick-17B-128E-Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2025-04-06",
      "observedAtMin": "2025-04-06",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-mmlu_external",
        "epoch-trivia_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www-cdn.anthropic.com/5c49cc247484cecf107c699baf29250302e5da70/claude-2-model-card.pdf",
          "line": 3652,
          "metricId": "EM",
          "observedAt": "2023-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=21;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-trivia_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://www-cdn.anthropic.com/5c49cc247484cecf107c699baf29250302e5da70/claude-2-model-card.pdf",
          "line": 5069,
          "metricId": "EM",
          "observedAt": "2023-04-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.867",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "trivia_qa_external.csv:row=5;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 86.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "claude-1.3",
      "numericRowCount": 2,
      "observedAtMax": "2023-04-18",
      "observedAtMin": "2023-04-18",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-simplebench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2244,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=794;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-simplebench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4647,
          "metricId": "Score (AVG@5)",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.464",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "simplebench_external.csv:row=61;column=Score (AVG@5)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.464
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Score (AVG@5)"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_12K",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-geobench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2249,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=801;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-geobench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://geobench.org/",
          "line": 2657,
          "metricId": "ACW Avg Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "3844",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "geobench_external.csv:row=15;column=ACW Avg Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 3844.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ACW Avg Score",
        "ECI Score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_15K",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://drb.futuresearch.ai",
          "line": 1600,
          "metricId": "Average score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.436",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=33;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 43.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2259,
          "metricId": "ECI Score",
          "observedAt": "2025-02-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=812;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average score",
        "ECI Score"
      ],
      "modelRef": "claude-3-7-sonnet-20250219_2K",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-mystery_game_puzzles"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1838,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=130;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.44
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mystery_game_puzzles",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3900,
          "metricId": "mean_score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.21",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mystery_game_puzzles.csv:row=47;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 21.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "mean_score"
      ],
      "modelRef": "claude-opus-4-1-20250805_24K",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-simplebench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2245,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=796;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-simplebench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4653,
          "metricId": "Score (AVG@5)",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.455",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "simplebench_external.csv:row=67;column=Score (AVG@5)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.455
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Score (AVG@5)"
      ],
      "modelRef": "claude-sonnet-4-20250514_12K",
      "numericRowCount": 2,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://drb.futuresearch.ai",
          "line": 1595,
          "metricId": "Average score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.466",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=28;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.6
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2256,
          "metricId": "ECI Score",
          "observedAt": "2025-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "142.32",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=809;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 142.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average score",
        "ECI Score"
      ],
      "modelRef": "claude-sonnet-4-20250514_2K",
      "numericRowCount": 2,
      "observedAtMax": "2025-05-22",
      "observedAtMin": "2025-05-22",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 453,
          "metricId": "Score",
          "observedAt": "2025-05-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.012700000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=177;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.012700000000000001
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 671,
          "metricId": "Score",
          "observedAt": "2025-05-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2733",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=182;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.2733
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "codex-mini-2025-05-16",
      "numericRowCount": 2,
      "observedAtMax": "2025-05-16",
      "observedAtMin": "2025-05-16",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1500,
          "metricId": "Accuracy",
          "observedAt": "2025-11-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=139;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4518,
          "metricId": "Score",
          "observedAt": "2025-11-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.409722222222222",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=86;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.409722222222222
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "cogito-671b-v2.1",
      "numericRowCount": 2,
      "observedAtMax": "2025-11-19",
      "observedAtMin": "2025-11-19",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1462,
          "metricId": "Accuracy",
          "observedAt": "2026-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00285714285714286",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=101;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.28571428571428603
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4524,
          "metricId": "Score",
          "observedAt": "2026-05-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.378472222222222",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=92;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.378472222222222
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "command-a-plus-05-2026",
      "numericRowCount": 2,
      "observedAtMax": "2026-05-20",
      "observedAtMin": "2026-05-20",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3560,
          "metricId": "mean_score",
          "observedAt": "2024-03-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.11650302114803625",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=93;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.650302114803624
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2924,
          "metricId": "mean_score",
          "observedAt": "2024-03-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.32891414141414144",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=250;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.891414141414145
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "dbrx-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2024-03-27",
      "observedAtMin": "2024-03-27",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3558,
          "metricId": "mean_score",
          "observedAt": "2023-11-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.06391616314199396",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=91;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.391616314199395
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2920,
          "metricId": "mean_score",
          "observedAt": "2023-11-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.24621212121212122",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=246;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 24.62121212121212
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "deepseek-llm-67b-chat",
      "numericRowCount": 2,
      "observedAtMax": "2023-11-29",
      "observedAtMin": "2023-11-29",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1484,
          "metricId": "Accuracy",
          "observedAt": "2025-12-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=123;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4532,
          "metricId": "Score",
          "observedAt": "2025-12-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.288194444444444",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=100;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.288194444444444
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "devstral-small-2512",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-09",
      "observedAtMin": "2025-12-09",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2218,
          "metricId": "ECI Score",
          "observedAt": "2023-05-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "103.88",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=745;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 103.88
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2949,
          "metricId": "EM",
          "observedAt": "2023-05-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.338",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=12;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 33.800000000000004
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "EM"
      ],
      "modelRef": "falcon-40b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2023-05-25",
      "observedAtMin": "2023-05-25",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2269,
          "metricId": "ECI Score",
          "observedAt": "2024-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "122.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=825;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 122.5
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3692,
          "metricId": "EM",
          "observedAt": "2024-05-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=64;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 77.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "EM"
      ],
      "modelRef": "gemini-1.5-flash-0514",
      "numericRowCount": 2,
      "observedAtMax": "2024-05-14",
      "observedAtMin": "2024-05-14",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-otis_mock_aime_2024_2025",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-otis_mock_aime_2024_2025",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4226,
          "metricId": "mean_score",
          "observedAt": "2024-10-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.04583333333333333",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "otis_mock_aime_2024_2025.csv:row=219;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 4.583333333333333
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2862,
          "metricId": "mean_score",
          "observedAt": "2024-10-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.32954545454545453",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=188;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.95454545454545
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "gemini-1.5-flash-8b-001",
      "numericRowCount": 2,
      "observedAtMax": "2024-10-03",
      "observedAtMin": "2024-10-03",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-enigma_eval_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1710,
          "metricId": "Accuracy",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0063",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=41;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.63
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2235,
          "metricId": "ECI Score",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "135.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=779;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 135.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "gemini-2.0-flash-02-05",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-05",
      "observedAtMin": "2025-02-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-frontiermath_tier_4",
        "frontiermath"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2566,
          "metricId": "mean_score",
          "observedAt": "2025-08-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.104",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=73;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 10.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "frontiermath",
          "benchmarkName": "FrontierMath T1–T3",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2494,
          "metricId": "mean_score",
          "observedAt": "2025-08-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath.csv:row=102;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 28.999999999999996
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "gemini-2.5-deep-think-2025-08-01-webapp",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-01",
      "observedAtMin": "2025-08-01",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-ale_bench_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 188,
          "metricId": "Performance",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "325.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=102;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 325.9
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5528,
          "metricId": "Accuracy",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3522",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=135;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.22
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Performance"
      ],
      "modelRef": "gemini-2.5-flash-lite-preview-06-17-thinking",
      "numericRowCount": 2,
      "observedAtMax": "2025-06-17",
      "observedAtMin": "2025-06-17",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2264,
          "metricId": "ECI Score",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=817;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.84
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5496,
          "metricId": "Accuracy",
          "observedAt": "2025-04-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.40950000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=103;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.95
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-04-17 (16K thinking)",
      "numericRowCount": 2,
      "observedAtMax": "2025-04-17",
      "observedAtMin": "2025-04-17",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2262,
          "metricId": "ECI Score",
          "observedAt": "2025-09-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "143.38",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=815;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 143.38
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5491,
          "metricId": "Accuracy",
          "observedAt": "2025-09-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4191",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=98;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.91
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "gemini-2.5-flash-preview-09-2025_16k",
      "numericRowCount": 2,
      "observedAtMax": "2025-09-25",
      "observedAtMin": "2025-09-25",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 31,
          "metricId": "Percent correct",
          "observedAt": "2025-06-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "83.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=18;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 83.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2103,
          "metricId": "ECI Score",
          "observedAt": "2025-06-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "145.84",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=493;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 145.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Percent correct"
      ],
      "modelRef": "gemini-2.5-pro-preview-06-05_32K",
      "numericRowCount": 2,
      "observedAtMax": "2025-06-06",
      "observedAtMin": "2025-06-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-deepresearchbench_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-deepresearchbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1596,
          "metricId": "Average score",
          "observedAt": "2025-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.46299999999999997",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "deepresearchbench_external.csv:row=29;column=Average score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.3
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2257,
          "metricId": "ECI Score",
          "observedAt": "2025-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "153.06",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=810;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 153.06
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average score",
        "ECI Score"
      ],
      "modelRef": "gemini-3-pro-preview_low",
      "numericRowCount": 2,
      "observedAtMax": "2025-11-18",
      "observedAtMin": "2025-11-18",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-algotune_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 206,
          "metricId": "Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.52",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=10;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.52
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5499,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4058",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=106;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.58
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "glm-4.5_thinking",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2265,
          "metricId": "ECI Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=818;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5501,
          "metricId": "Accuracy",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.39770000000000005",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=108;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.77
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "gpt-5-chat",
      "numericRowCount": 2,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-apex_agents_external",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 249,
          "metricId": "Pass@1 score",
          "observedAt": "2025-09-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.201",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=35;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 20.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5449,
          "metricId": "Accuracy",
          "observedAt": "2025-09-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.5453",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=56;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.53
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Pass@1 score"
      ],
      "modelRef": "gpt-5-codex_high",
      "numericRowCount": 2,
      "observedAtMax": "2025-09-15",
      "observedAtMin": "2025-09-15",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-algotune_external",
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-algotune_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 215,
          "metricId": "Score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1.31",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "algotune_external.csv:row=19;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 1.31
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2214,
          "metricId": "ECI Score",
          "observedAt": "2025-10-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "150.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=695;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 150.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Score"
      ],
      "modelRef": "gpt-5-pro-2025-10-06_medium",
      "numericRowCount": 2,
      "observedAtMax": "2025-10-07",
      "observedAtMin": "2025-10-07",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2263,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=816;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5492,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.41880000000000006",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=99;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.88000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "gpt-oss-120b_medium",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2267,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "136.81",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=821;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 136.81
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5522,
          "metricId": "Accuracy",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3683",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=129;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 36.83
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "gpt-oss-20b_medium",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1492,
          "metricId": "Accuracy",
          "observedAt": "2026-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=131;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4537,
          "metricId": "Score",
          "observedAt": "2026-04-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.258101851851852",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=105;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.258101851851852
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "granite-4.1-30b",
      "numericRowCount": 2,
      "observedAtMax": "2026-04-29",
      "observedAtMin": "2026-04-29",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2261,
          "metricId": "ECI Score",
          "observedAt": "2025-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=814;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.04
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5489,
          "metricId": "Accuracy",
          "observedAt": "2025-06-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4258",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=96;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 42.58
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "grok-3-mini_high",
      "numericRowCount": 2,
      "observedAtMax": "2025-06-24",
      "observedAtMin": "2025-06-24",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-apex_agents_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-apex_agents_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 262,
          "metricId": "Pass@1 score",
          "observedAt": "2025-11-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.128",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "apex_agents_external.csv:row=48;column=Pass@1 score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 12.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-webdev_arena_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5391,
          "metricId": "Arena Score",
          "observedAt": "2025-11-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1209.85",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "webdev_arena_external.csv:row=119;column=Arena Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "elo",
          "value": 1209.85
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Arena Score",
        "Pass@1 score"
      ],
      "modelRef": "grok-4-1",
      "numericRowCount": 2,
      "observedAtMax": "2025-11-17",
      "observedAtMin": "2025-11-17",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2266,
          "metricId": "ECI Score",
          "observedAt": "2025-07-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.58",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=819;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.58
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5506,
          "metricId": "Accuracy",
          "observedAt": "2025-07-12",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3936",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=113;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 39.36
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "ECI Score"
      ],
      "modelRef": "kimi-k2-0711-preview",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-12",
      "observedAtMin": "2025-07-12",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-proofbench_external",
        "epoch-webdev_arena_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-proofbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4418,
          "metricId": "Accuracy",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "proofbench_external.csv:row=58;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-webdev_arena_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 5377,
          "metricId": "Arena Score",
          "observedAt": "2026-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "1301.97",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "webdev_arena_external.csv:row=103;column=Arena Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "elo",
          "value": 1301.97
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Arena Score"
      ],
      "modelRef": "laguna-xs.2",
      "numericRowCount": 2,
      "observedAtMax": "2026-04-28",
      "observedAtMin": "2026-04-28",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4573,
          "metricId": "Score",
          "observedAt": "2024-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.706",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=86;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.706
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 4580,
          "metricId": "Score",
          "observedAt": "2024-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.701",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=93;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.701
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "llava-v1.6-vicuna-7b",
      "numericRowCount": 2,
      "observedAtMax": "2024-01-31",
      "observedAtMin": "2024-01-31",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1460,
          "metricId": "Accuracy",
          "observedAt": "2025-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00285714285714286",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=99;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.28571428571428603
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4522,
          "metricId": "Score",
          "observedAt": "2025-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.392361111111111",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=90;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.392361111111111
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "magistral-medium-2509",
      "numericRowCount": 2,
      "observedAtMax": "2025-09-18",
      "observedAtMin": "2025-09-18",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1461,
          "metricId": "Accuracy",
          "observedAt": "2025-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.00285714285714286",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=100;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.28571428571428603
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4529,
          "metricId": "Score",
          "observedAt": "2025-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.351851851851852",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=97;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.351851851851852
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "magistral-small-2509",
      "numericRowCount": 2,
      "observedAtMax": "2025-09-18",
      "observedAtMin": "2025-09-18",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3562,
          "metricId": "mean_score",
          "observedAt": "2024-10-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.14444864048338368",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=95;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.444864048338369
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2926,
          "metricId": "mean_score",
          "observedAt": "2024-10-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.25252525252525254",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=252;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 25.252525252525253
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "ministral-3b-2410",
      "numericRowCount": 2,
      "observedAtMax": "2024-10-16",
      "observedAtMin": "2024-10-16",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3564,
          "metricId": "mean_score",
          "observedAt": "2024-10-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.14935800604229607",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=97;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 14.935800604229607
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2929,
          "metricId": "mean_score",
          "observedAt": "2024-10-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.27146464646464646",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=255;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.146464646464647
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "ministral-8b-2410",
      "numericRowCount": 2,
      "observedAtMax": "2024-10-16",
      "observedAtMin": "2024-10-16",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1493,
          "metricId": "Accuracy",
          "observedAt": "2025-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=132;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4535,
          "metricId": "Score",
          "observedAt": "2025-06-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.263888888888889",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=103;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.263888888888889
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "mistral-small-2506",
      "numericRowCount": 2,
      "observedAtMax": "2025-06-20",
      "observedAtMin": "2025-06-20",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "aider-polyglot",
        "epoch-ale_bench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 84,
          "metricId": "Percent correct",
          "observedAt": "2025-07-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=75;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.1
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 190,
          "metricId": "Performance",
          "observedAt": "2025-09-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "267.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=104;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 267.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Percent correct",
        "Performance"
      ],
      "modelRef": "moonshotai/kimi-k2-0905",
      "numericRowCount": 2,
      "observedAtMax": "2025-09-05",
      "observedAtMin": "2025-07-17",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-surface_evolver_bench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2056,
          "metricId": "ECI Score",
          "observedAt": "2026-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "154.33",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=400;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 154.33
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-surface_evolver_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4811,
          "metricId": "Mean score",
          "observedAt": "2026-07-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.525",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "surface_evolver_bench_external.csv:row=25;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Mean score"
      ],
      "modelRef": "muse-spark-1.1_high",
      "numericRowCount": 2,
      "observedAtMax": "2026-07-09",
      "observedAtMin": "2026-07-09",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1498,
          "metricId": "Accuracy",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=137;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4559,
          "metricId": "Score",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.38657407407407407",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=127;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.38657407407407407
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "nova-2.0-pro-preview_low",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-02",
      "observedAtMin": "2025-12-02",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1486,
          "metricId": "Accuracy",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=125;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4515,
          "metricId": "Score",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.42708333333333304",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=83;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.42708333333333304
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "nova-2.0-pro-preview_medium",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-02",
      "observedAtMin": "2025-12-02",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1497,
          "metricId": "Accuracy",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=136;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4533,
          "metricId": "Score",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.28125",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=101;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.28125
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "nova-2.0-pro-preview_none",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-02",
      "observedAtMin": "2025-12-02",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-enigma_eval_external",
        "hle"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1688,
          "metricId": "Accuracy",
          "observedAt": "2025-03-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.061399999999999996",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=19;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.14
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3294,
          "metricId": "Accuracy",
          "observedAt": "2025-03-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0812",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "hle_external.csv:row=34;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy"
      ],
      "modelRef": "o1-pro-2025-03-19",
      "numericRowCount": 2,
      "observedAtMax": "2025-03-19",
      "observedAtMin": "2025-03-19",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2155,
          "metricId": "ECI Score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.49",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=600;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.49
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2321,
          "metricId": "Overall score",
          "observedAt": "2025-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "59.6",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=38;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 59.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Overall score"
      ],
      "modelRef": "o3-mini-2025-01-31_unknown",
      "numericRowCount": 2,
      "observedAtMax": "2025-01-31",
      "observedAtMin": "2025-01-31",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index",
        "epoch-forecastbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2150,
          "metricId": "ECI Score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "146.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=594;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 146.4
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2289,
          "metricId": "Overall score",
          "observedAt": "2025-04-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "61.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=6;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 61.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "ECI Score",
        "Overall score"
      ],
      "modelRef": "o4-mini-2025-04-16_unknown",
      "numericRowCount": 2,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-math_level_5",
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-math_level_5",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3563,
          "metricId": "mean_score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0368202416918429",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "math_level_5.csv:row=96;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.68202416918429
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2932,
          "metricId": "mean_score",
          "observedAt": "2023-09-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.13226010101010102",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=258;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 13.226010101010102
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "open-mistral-7b",
      "numericRowCount": 2,
      "observedAtMax": "2023-09-27",
      "observedAtMin": "2023-09-27",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-balrog_external",
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 894,
          "metricId": "Average progress",
          "observedAt": "2025-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.078",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=35;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 7.8
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3829,
          "metricId": "EM",
          "observedAt": "2025-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.729",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=222;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 72.89999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Average progress",
        "EM"
      ],
      "modelRef": "qwen2.5-7b-instruct",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-26",
      "observedAtMin": "2025-02-26",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-critpt_external",
        "epoch-scicode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-critpt_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1489,
          "metricId": "Accuracy",
          "observedAt": "2026-01-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "critpt_external.csv:row=128;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.0
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-scicode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4538,
          "metricId": "Score",
          "observedAt": "2026-01-27",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.246527777777778",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "scicode_external.csv:row=106;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.246527777777778
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Accuracy",
        "Score"
      ],
      "modelRef": "solar-pro-3",
      "numericRowCount": 2,
      "observedAtMax": "2026-01-27",
      "observedAtMin": "2026-01-27",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-arc_agi_2_external",
        "epoch-arc_agi_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_2_external",
          "benchmarkName": "ARC-AGI-2 (Epoch external aggregation)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 398,
          "metricId": "Score",
          "observedAt": "2025-10-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0625",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_2_external.csv:row=122;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0625
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 644,
          "metricId": "Score",
          "observedAt": "2025-10-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=154;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "tiny-recursion-model",
      "numericRowCount": 2,
      "observedAtMax": "2025-10-06",
      "observedAtMin": "2025-10-06",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "epoch-bool_q_external",
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-bool_q_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/boolq",
          "line": 1041,
          "metricId": "Score",
          "observedAt": "2023-06-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.808",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "bool_q_external.csv:row=44;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.808
        },
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2953,
          "metricId": "EM",
          "observedAt": "2023-06-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.226",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=16;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 22.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "EM",
        "Score"
      ],
      "modelRef": "vicuna-13b-v1.3",
      "numericRowCount": 2,
      "observedAtMax": "2023-06-18",
      "observedAtMin": "2023-06-18",
      "rowCount": 2,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-multimodal"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 362,
          "metricId": "resolved",
          "observedAt": "2024-10-06",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Multimodal",
            "scaffold": "SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 12.19,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[13]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 12.19
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 363,
          "metricId": "resolved",
          "observedAt": "2024-10-06",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent JavaScript",
            "leaderboard_variant": "Multimodal",
            "scaffold": "SWE-agent JavaScript",
            "subject_type": "system"
          },
          "rawValue": 11.99,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[14]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 11.99
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude Sonnet 3.5",
      "numericRowCount": 2,
      "observedAtMax": "2024-10-06",
      "observedAtMin": "2024-10-06",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 33,
          "metricId": "resolved",
          "observedAt": "2025-12-09",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 53.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[32]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 53.8
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 180,
          "metricId": "resolved",
          "observedAt": "2025-12-09",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 53.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[95]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 53.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Devstral (2512)",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-09",
      "observedAtMin": "2025-12-09",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 28,
          "metricId": "resolved",
          "observedAt": "2025-12-09",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 56.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[27]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 56.4
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 172,
          "metricId": "resolved",
          "observedAt": "2025-12-09",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 56.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[87]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 56.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Devstral Small (2512)",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-09",
      "observedAtMin": "2025-12-09",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 32,
          "metricId": "resolved",
          "observedAt": "2025-08-22",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 54.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[31]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 54.2
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 179,
          "metricId": "resolved",
          "observedAt": "2025-08-22",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 54.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[94]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 54.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GLM 4.5",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-22",
      "observedAtMin": "2025-08-22",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 31,
          "metricId": "resolved",
          "observedAt": "2025-12-01",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 55.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[30]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 55.4
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 176,
          "metricId": "resolved",
          "observedAt": "2025-12-01",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 55.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[91]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 55.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GLM 4.6",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-01",
      "observedAtMin": "2025-12-01",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 38,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 39.58,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[37]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 39.58
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 224,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 39.58,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[139]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 39.58
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT 4.1",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-26",
      "observedAtMin": "2025-07-26",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 42,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 23.94,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[41]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 23.94
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 247,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 23.94,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[162]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 23.94
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT 4.1 mini",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-20",
      "observedAtMin": "2025-07-20",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 43,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 21.62,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[42]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 21.62
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 251,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 21.62,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[166]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 21.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT 4o",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-20",
      "observedAtMin": "2025-07-20",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 45,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 13.52,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[44]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 13.52
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 255,
          "metricId": "resolved",
          "observedAt": "2025-07-26",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 13.52,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[170]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 13.52
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Gemini 2.0 Flash",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-26",
      "observedAtMin": "2025-07-26",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 119,
          "metricId": "resolved",
          "observedAt": "2025-10-14",
          "protocol": {
            "checked": false,
            "harness": "Lingxi v1.5",
            "leaderboard_variant": "Verified",
            "scaffold": "Lingxi v1.5",
            "subject_type": "model"
          },
          "rawValue": 71.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[34]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 71.2
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 145,
          "metricId": "resolved",
          "observedAt": "2025-07-16",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Verified",
            "scaffold": "OpenHands",
            "subject_type": "model"
          },
          "rawValue": 65.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[60]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 65.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Kimi K2",
      "numericRowCount": 2,
      "observedAtMax": "2025-10-14",
      "observedAtMin": "2025-07-16",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 23,
          "metricId": "resolved",
          "observedAt": "2025-12-10",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 63.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[22]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 63.4
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 152,
          "metricId": "resolved",
          "observedAt": "2025-12-10",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 63.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[67]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 63.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Kimi K2 Thinking",
      "numericRowCount": 2,
      "observedAtMax": "2025-12-10",
      "observedAtMin": "2025-12-10",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 44,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 21.04,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[43]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 21.04
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 252,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 21.04,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[167]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 21.04
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Llama 4 Maverick Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-20",
      "observedAtMin": "2025-07-20",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 46,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 9.06,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[45]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 9.06
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 257,
          "metricId": "resolved",
          "observedAt": "2025-07-20",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 9.06,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[172]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 9.06
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Llama 4 Scout Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2025-07-20",
      "observedAtMin": "2025-07-20",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 341,
          "metricId": "resolved",
          "observedAt": "2025-06-27",
          "protocol": {
            "checked": false,
            "harness": "MCTS-Refine-7B",
            "leaderboard_variant": "Lite",
            "scaffold": "MCTS-Refine-7B",
            "subject_type": "model"
          },
          "rawValue": 16.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[76]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 16.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 248,
          "metricId": "resolved",
          "observedAt": "2025-06-27",
          "protocol": {
            "checked": false,
            "harness": "MCTS-Refine-7B",
            "leaderboard_variant": "Verified",
            "scaffold": "MCTS-Refine-7B",
            "subject_type": "model"
          },
          "rawValue": 23.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[163]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 23.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "MCTS Refine 7B",
      "numericRowCount": 2,
      "observedAtMax": "2025-06-27",
      "observedAtMin": "2025-06-27",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 24,
          "metricId": "resolved",
          "observedAt": "2025-11-24",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 61.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[23]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 61.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 158,
          "metricId": "resolved",
          "observedAt": "2025-11-24",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 61.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[73]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 61.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "MiniMax M2",
      "numericRowCount": 2,
      "observedAtMax": "2025-11-24",
      "observedAtMin": "2025-11-24",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-lite",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 308,
          "metricId": "resolved",
          "observedAt": "2025-02-14",
          "protocol": {
            "checked": true,
            "harness": "Agentless Lite",
            "leaderboard_variant": "Lite",
            "scaffold": "Agentless Lite",
            "subject_type": "system"
          },
          "rawValue": 32.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[43]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 32.33
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 212,
          "metricId": "resolved",
          "observedAt": "2025-02-14",
          "protocol": {
            "checked": true,
            "harness": "Agentless Lite",
            "leaderboard_variant": "Verified",
            "scaffold": "Agentless Lite",
            "subject_type": "system"
          },
          "rawValue": 42.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[127]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 42.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "O3 Mini",
      "numericRowCount": 2,
      "observedAtMax": "2025-02-14",
      "observedAtMin": "2025-02-14",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 47,
          "metricId": "resolved",
          "observedAt": "2025-08-03",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 9.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[46]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 9.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 258,
          "metricId": "resolved",
          "observedAt": "2025-08-03",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 9.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[173]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 9.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Qwen2.5-Coder 32B Instruct",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-03",
      "observedAtMin": "2025-08-03",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 2,
      "benchmarkIds": [
        "swebench-bash-only",
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-bash-only",
          "benchmarkName": null,
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 41,
          "metricId": "resolved",
          "observedAt": "2025-08-07",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "bash-only",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 26.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[0]=bash-only;results[40]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 26.0
        },
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 242,
          "metricId": "resolved",
          "observedAt": "2025-08-07",
          "protocol": {
            "checked": true,
            "harness": "mini-SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "mini-SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 26.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[157]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 26.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 2
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "gpt-oss-120b",
      "numericRowCount": 2,
      "observedAtMax": "2025-08-07",
      "observedAtMin": "2025-08-07",
      "rowCount": 2,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 2
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 5,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 96.4,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=5;row=4",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 96.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/LGAI-EXAONE/EXAONE-4.5-33B",
          "line": 18,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 92.6,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=18;row=17",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 92.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/EXAONE-4.5-33B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "line": 21,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87.5,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=21;row=20",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 87.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-30B-A3B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "line": 23,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.5,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=23;row=22",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 82.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-4B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-2.6-flash",
          "line": 24,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 73.85,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=24;row=23",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 73.85
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-2.6-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-3.0-flash",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 93.2,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 93.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-3.0-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ring-2.6-1T",
          "line": 8,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 95.83,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=8;row=7",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 95.83
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ring-2.6-1T",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
          "line": 12,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Muse-Glimmer-30B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 94.7,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=12;row=11",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 94.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-models/Muse-Glimmer-30B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "line": 20,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 90,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=20;row=19",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 90.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/skt/A.X-K2/blob/fb412dcd22e9c5ac43b2967d6c5e8a7d0e667b0f/README.md#L134-L152",
          "line": 2,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 97.1,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=2;row=1",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 97.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "skt/A.X-K2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "line": 3,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 96.67,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=3;row=2",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 96.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling",
          "line": 1,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 97.1,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=1;row=0",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 97.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling-Small",
          "line": 10,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 95.5,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=10;row=9",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 95.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling-Small",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aime-2026"
      ],
      "examples": [
        {
          "artifact": "hf-aime-2026/candidates.jsonl",
          "benchmarkId": "aime-2026",
          "benchmarkName": "AIME 2026",
          "evidenceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "line": 9,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/aime_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 95.7,
          "sourceId": "hf-aime-2026",
          "sourceLabel": "Hugging Face · AIME leaderboard API",
          "sourceLocator": "rank=9;row=8",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard",
          "unit": "percent",
          "value": 95.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "upstage/Solar-Open2-250B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-aime-2026",
      "sourceIds": [
        "hf-aime-2026"
      ],
      "sourceLabels": [
        "Hugging Face · AIME leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/aime_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b2/77/b27740dc-a017-4738-8dd0-c18973b61d9e.json",
          "line": 102,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 50.5050505051,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=102;row=101",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 50.5050505051
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/c4ai-command-a-03-2025",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9a/17/9a17fcf2-ee22-4a4f-b992-0d07290a56fc.json",
          "line": 113,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 26.7676767677,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=113;row=112",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 26.7676767677
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/c4ai-command-r-08-2024",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/af/03/af03cc06-a493-48b7-a049-61d9b690c8f3.json",
          "line": 109,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 34.3434343434,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=109;row=108",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 34.3434343434
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/c4ai-command-r-plus-08-2024",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/69/1d/691d4ef0-94b0-47e2-a7c9-b85f1bdc625b.json",
          "line": 114,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 26.7676767677,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=114;row=113",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 26.7676767677
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/c4ai-command-r7b-12-2024",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-27B-Opus",
          "line": 28,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=28;row=27",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 86.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-27B-Opus",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-28B-REASON",
          "line": 12,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 89.39,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=12;row=11",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 89.39
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-28B-REASON",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-31B-Opus",
          "line": 36,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 85.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=36;row=35",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 85.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-31B-Opus",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-36B-Opus",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 88.4,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 88.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-36B-Opus",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-398B-JGOS",
          "line": 6,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 90.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=6;row=5",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 90.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-398B-JGOS",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-4B-David",
          "line": 42,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 85,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=42;row=41",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 85.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-4B-David",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-60B-DUO",
          "line": 17,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 88.38,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=17;row=16",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 88.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-60B-DUO",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-9B-NEG",
          "line": 45,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.34,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=45;row=44",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 84.34
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Darwin-9B-NEG",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Ourbox-35B-JGOS",
          "line": 31,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.36,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=31;row=30",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 86.36
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FINAL-Bench/Ourbox-35B-JGOS",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-9B-NEG",
          "line": 47,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.34,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=47;row=46",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 84.34
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FlagRelease/Darwin-9B-NEG-FINAL-hygon-FlagOS",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/FINAL-Bench/Darwin-9B-NEG",
          "line": 48,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.34,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=48;row=47",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 84.34
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "FlagRelease/Darwin-9B-NEG-ansulev-hygon-FlagOS",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 8,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 90.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=8;row=7",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 90.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/JGOS-Model/JGOS-31B-Citizen",
          "line": 46,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.34,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=46;row=45",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 84.34
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JGOS-Model/JGOS-31B-Citizen",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "line": 112,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31.31,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=112;row=111",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 31.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "line": 111,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31.31,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=111;row=110",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 31.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Base-Pretrain",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "line": 106,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 40.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=106;row=105",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 40.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "line": 108,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 38.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=108;row=107",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 38.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Instruct-SFT",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "line": 100,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 57.6,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=100;row=99",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 57.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Thinking",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "line": 107,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 39.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=107;row=106",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 39.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Thinking-SFT",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/LGAI-EXAONE/EXAONE-4.5-33B",
          "line": 65,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=65;row=64",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 80.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/EXAONE-4.5-33B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/LGAI-EXAONE/K-EXAONE-236B-A23B",
          "line": 73,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79.1,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=73;row=72",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 79.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/K-EXAONE-236B-A23B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4b/02/4b02c12c-3f7b-45c8-b746-c590860a7a36.json",
          "line": 85,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 72.2222222222,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=85;row=84",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 72.2222222222
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-30B-A3B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/Qwen/Qwen3-4B-Instruct-2507",
          "line": 96,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 62,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=96;row=95",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 62.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-4B-Instruct-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/Qwen/Qwen3-4B-Thinking-2507",
          "line": 95,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 65.8,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=95;row=94",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 65.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-4B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/deepseek-ai/DeepSeek-R1",
          "line": 87,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=87;row=86",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 71.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-R1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0a/db/0adbcd9f-38e6-414a-88b1-7bd928d4c72e.json",
          "line": 98,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 58.2070707071,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=98;row=97",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 58.2070707071
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-V3",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/ibm-granite/granite-4.1-30b",
          "line": 104,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45.76,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=104;row=103",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 45.76
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-4.1-30b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/ibm-granite/granite-4.1-3b",
          "line": 110,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31.7,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=110;row=109",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 31.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-4.1-3b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/ibm-granite/granite-4.1-8b",
          "line": 105,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 41.96,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=105;row=104",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 41.96
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-4.1-8b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ring-2.6-1T",
          "line": 18,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 88.27,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=18;row=17",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 88.27
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ring-2.6-1T",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/jdopensource/JoyAI-LLM-Flash",
          "line": 80,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74.43,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=80;row=79",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 74.43
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "jdopensource/JoyAI-LLM-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/meituan-longcat/LongCat-Flash-Thinking-2601",
          "line": 64,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=64;row=63",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 80.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meituan-longcat/LongCat-Flash-Thinking-2601",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/aa/ad/aaad0db3-7c5b-4834-a1a2-d96624bdd003.json",
          "line": 115,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.6868686869,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=115;row=114",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 18.6868686869
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.2-1B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/94/0f/940ff5c0-b932-4c37-9eb5-27cbd4997484.json",
          "line": 103,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 46.0858585859,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=103;row=102",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 46.0858585859
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.2-90B-Vision-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
          "line": 51,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Muse-Glimmer-30B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=51;row=50",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 83.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-models/Muse-Glimmer-30B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603",
          "line": 89,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.2,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=89;row=88",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 71.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Small-4-119B-2603",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2-Thinking",
          "line": 44,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=44;row=43",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 84.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Thinking",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/bd/1f/bd1fe04f-7d58-4045-be60-7150500d537f.json",
          "line": 75,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 77.2727272727,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=75;row=74",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 77.2727272727
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
          "line": 27,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=27;row=26",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 87.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
          "line": 20,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87.9,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=20;row=19",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 87.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
          "line": 78,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 75.44,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=78;row=77",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 75.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "line": 77,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 75.57,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=77;row=76",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 75.57
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/nvidia/Nemotron-Cascade-2-30B-A3B",
          "line": 76,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 76.1,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=76;row=75",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 76.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/Nemotron-Cascade-2-30B-A3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-35B-A3B",
          "line": 14,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-35b-a3b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 89.2,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=14;row=13",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 89.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-35B-A3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-397B",
          "line": 2,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-397b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 92.8,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=2;row=1",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 92.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-397B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-9B",
          "line": 30,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-9b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.4,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=30;row=29",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 86.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-9B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/skt/A.X-K1",
          "line": 82,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/A.X-K1.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=82;row=81",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 74.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "skt/A.X-K1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/skt/A.X-K2/blob/fb412dcd22e9c5ac43b2967d6c5e8a7d0e667b0f/README.md#L134-L152",
          "line": 38,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 85.6,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=38;row=37",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 85.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "skt/A.X-K2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://arxiv.org/abs/2602.10604",
          "line": 50,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa_diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=50;row=49",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 83.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3",
          "line": 9,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 90.4,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=9;row=8",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 90.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3-preview",
          "line": 25,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87.2,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=25;row=24",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 87.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3-preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling",
          "line": 26,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87.2,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=26;row=25",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 87.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling-Small",
          "line": 11,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 89.5,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=11;row=10",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 89.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling-Small",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "line": 32,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.3,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=32;row=31",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 86.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "upstage/Solar-Open2-250B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9c/77/9c7740fd-9cad-445c-9681-eb576e2a110e.json",
          "line": 86,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.7171717172,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=86;row=85",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 71.7171717172
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.5-Air",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "hf-gpqa/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/7e/18/7e18ad63-4a0a-4d9d-a4ec-40b683b42981.json",
          "line": 60,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/gpqa-diamond.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.3131313131,
          "sourceId": "hf-gpqa",
          "sourceLabel": "Hugging Face · GPQA leaderboard API",
          "sourceLocator": "rank=60;row=59",
          "sourceUrl": "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard",
          "unit": "percent",
          "value": 81.3131313131
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.6-FP8",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-gpqa",
      "sourceIds": [
        "hf-gpqa"
      ],
      "sourceLabels": [
        "Hugging Face · GPQA leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/Idavidrein/gpqa/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/HelpingAI/Dhanishtha-2.0-0126",
          "line": 80,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 9.92,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=80;row=79",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 9.92
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "HelpingAI/Dhanishtha-2.0-0126",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/InternScience/Agents-A1",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Agents-A1.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 47.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 47.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "InternScience/Agents-A1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/LGAI-EXAONE/K-EXAONE-236B-A23B",
          "line": 74,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 13.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=74;row=73",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 13.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/K-EXAONE-236B-A23B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M2",
          "line": 75,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 12.5,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=75;row=74",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 12.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MiniMaxAI/MiniMax-M2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/PolarSeeker/OpenSeeker-v2-30B-SFT",
          "line": 31,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 34.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=31;row=30",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 34.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "PolarSeeker/OpenSeeker-v2-30B-SFT",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/dots-studio/dots3-note-prev",
          "line": 8,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/dots3-note-prev.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=8;row=7",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 52.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "dots-studio/dots3-note-prev",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-3.0-flash",
          "line": 57,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.7,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=57;row=56",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-3.0-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/internlm/Intern-S2-Preview",
          "line": 61,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 21.94,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=61;row=60",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 21.94
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/Intern-S2-Preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
          "line": 60,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Muse-Glimmer-30B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=60;row=59",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 22.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-models/Muse-Glimmer-30B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
          "line": 76,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 11.72,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=76;row=75",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 11.72
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "line": 79,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 10.47,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=79;row=78",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 10.47
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/skt/A.X-K1",
          "line": 84,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/A.X-K1.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 8.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=84;row=83",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 8.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "skt/A.X-K1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/skt/A.X-K2/blob/fb412dcd22e9c5ac43b2967d6c5e8a7d0e667b0f/README.md#L134-L152",
          "line": 41,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 27.8,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=41;row=40",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 27.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "skt/A.X-K2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://arxiv.org/abs/2602.10604",
          "line": 51,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 23.1,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=51;row=50",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 23.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/stepfun-ai/Step-3.7-Flash",
          "line": 14,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle_with_tools.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 48.1,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=14;row=13",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 48.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.7-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3",
          "line": 7,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.2,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=7;row=6",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 53.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3-preview",
          "line": 38,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 30,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=38;row=37",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 30.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3-preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling",
          "line": 18,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 46,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=18;row=17",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 46.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling-Small",
          "line": 33,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31.6,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=33;row=32",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 31.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling-Small",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "line": 39,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 28.8,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=39;row=38",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 28.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "upstage/Solar-Open2-250B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/93/cb/93cbdfdf-9ec6-4afb-87f4-d771cf29a378.json",
          "line": 85,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 8.32,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=85;row=84",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 8.32
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.5",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hle"
      ],
      "examples": [
        {
          "artifact": "hf-hle/candidates.jsonl",
          "benchmarkId": "hle",
          "benchmarkName": "Humanity's Last Exam",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e9/03/e90308ab-e898-4434-aa93-13d131887737.json",
          "line": 86,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hle.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 8.12,
          "sourceId": "hf-hle",
          "sourceLabel": "Hugging Face · Humanity's Last Exam leaderboard API",
          "sourceLocator": "rank=86;row=85",
          "sourceUrl": "https://huggingface.co/api/datasets/cais/hle/leaderboard",
          "unit": "percent",
          "value": 8.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.5-Air",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hle",
      "sourceIds": [
        "hf-hle"
      ],
      "sourceLabels": [
        "Hugging Face · Humanity's Last Exam leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/cais/hle/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 2,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 92.7,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=2;row=1",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 92.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://matharena.ai/?comp=hmmt--hmmt_feb_2026",
          "line": 15,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 78.79,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=15;row=14",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 78.79
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-30B-A3B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://matharena.ai/?comp=hmmt--hmmt_feb_2026",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.03,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 53.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-4B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-2.6-flash",
          "line": 17,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 49.29,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=17;row=16",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 49.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-2.6-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-3.0-flash",
          "line": 7,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=7;row=6",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 87.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-3.0-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://huggingface.co/internlm/Intern-S2-Preview",
          "line": 5,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87.31,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=5;row=4",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 87.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/Intern-S2-Preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://matharena.ai/?comp=hmmt--hmmt_feb_2026",
          "line": 10,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.85,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=10;row=9",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 84.85
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://matharena.ai/?comp=hmmt--hmmt_feb_2026",
          "line": 8,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/MathArena--hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.36,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=8;row=7",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 86.36
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "hmmt"
      ],
      "examples": [
        {
          "artifact": "hf-hmmt-2026/candidates.jsonl",
          "benchmarkId": "hmmt",
          "benchmarkName": "HMMT",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling-Small",
          "line": 3,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/hmmt_feb_2026.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 90.2,
          "sourceId": "hf-hmmt-2026",
          "sourceLabel": "Hugging Face · HMMT leaderboard API",
          "sourceLocator": "rank=3;row=2",
          "sourceUrl": "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard",
          "unit": "percent",
          "value": 90.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling-Small",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-hmmt-2026",
      "sourceIds": [
        "hf-hmmt-2026"
      ],
      "sourceLabels": [
        "Hugging Face · HMMT leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/MathArena/hmmt_feb_2026/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/67/6f/676f4465-ce78-411a-9f5a-c97b3d2eac4f.json",
          "line": 80,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.29,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=80;row=79",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 52.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "01-ai/Yi-1.5-34B-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/bd/05/bd056a61-8ede-45b4-823d-093343bdd880.json",
          "line": 108,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 38.23,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=108;row=107",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 38.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "01-ai/Yi-1.5-6B-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/94/9e/949ec028-f03f-4211-8783-c810af6489a4.json",
          "line": 90,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45.95,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=90;row=89",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 45.95
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "01-ai/Yi-1.5-9B-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9d/42/9d42931c-6ec3-48ac-ae0a-cebb9c1f68aa.json",
          "line": 98,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 43.03,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=98;row=97",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 43.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "01-ai/Yi-34B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/8b/02/8b02ff6f-b3ae-4230-8edd-50c3ffa34592.json",
          "line": 128,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 26.51,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=128;row=127",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 26.51
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "01-ai/Yi-6B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/28/1d/281df2d2-60b5-4053-b30e-bba4976c3efd.json",
          "line": 127,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 28.84,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=127;row=126",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 28.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "01-ai/Yi-6B-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/cf/2d/cf2d98c9-1e57-4994-9e49-9a7962a2706d.json",
          "line": 57,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 65.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=57;row=56",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 65.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ByteDance-Seed/Seed-OSS-36B-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/60/1d/601d782d-660b-4f01-bcae-5c4105223c85.json",
          "line": 65,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=65;row=64",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 60.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ByteDance-Seed/Seed-OSS-36B-Base-woSyn",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/1b/40/1b405e01-7137-4189-b5d2-83e0f4004dc9.json",
          "line": 28,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=28;row=27",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 82.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ByteDance-Seed/Seed-OSS-36B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/12/29/1229c8e0-d21f-43e8-a9c7-3ea0d09408d5.json",
          "line": 92,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45.41,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=92;row=91",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 45.41
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/aya-expanse-32b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/33/5b/335b15a4-a7fd-4ffc-81db-9d5feed90657.json",
          "line": 119,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 33.74,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=119;row=118",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 33.74
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/aya-expanse-8b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b3/5f/b35fcd95-c193-4ed0-b50c-bc535af5756f.json",
          "line": 111,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.9,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=111;row=110",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 37.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/c4ai-command-r-v01",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/f0/87/f087fc33-d4b9-4939-b3cf-2d3fb9094177.json",
          "line": 130,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 23.45,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=130;row=129",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 23.45
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "EleutherAI/llemma_7b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/67/cc/67cc35fb-ae5a-4632-91da-9346a6b2849e.json",
          "line": 113,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=113;row=112",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 37.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "GSAI-ML/LLaDA-8B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/33/c9/33c9180c-621e-40a2-bb6f-02b90ecff1fd.json",
          "line": 136,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 18.31,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=136;row=135",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 18.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "HuggingFaceTB/SmolLM2-1.7B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ac/90/ac90804d-cb99-41cc-b3e7-65ed31484008.json",
          "line": 142,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 10.85,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=142;row=141",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 10.85
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "HuggingFaceTB/SmolLM2-135M",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/23/ae/23aee40b-5bea-46cf-a87b-8263e12e2d1f.json",
          "line": 141,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 11.38,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=141;row=140",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 11.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "HuggingFaceTB/SmolLM2-360M",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "line": 68,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.31,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=68;row=67",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 59.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "line": 67,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mellum2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.31,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=67;row=66",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 59.31
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "JetBrains/Mellum2-12B-A2.5B-Base-Pretrain",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ce/1a/ce1a8a1a-5d16-4403-9e85-bf0f3d72f09f.json",
          "line": 107,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 39.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=107;row=106",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 39.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/EXAONE-3.5-2.4B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/de/c9/dec9e4cf-cc67-42e1-9474-f8de9048014b.json",
          "line": 69,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 58.91,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=69;row=68",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 58.91
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/EXAONE-3.5-32B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/c6/8f/c68fc694-d6dd-4b2a-a699-e5f737ae550f.json",
          "line": 89,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 46.24,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=89;row=88",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 46.24
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/LGAI-EXAONE/EXAONE-4.5-33B",
          "line": 26,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.3,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=26;row=25",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/EXAONE-4.5-33B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/LGAI-EXAONE/K-EXAONE-236B-A23B",
          "line": 19,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.8,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=19;row=18",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "LGAI-EXAONE/K-EXAONE-236B-A23B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/3b/7b/3b7bfaac-ee52-4508-8ecd-676d82e06c1c.json",
          "line": 36,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=36;row=35",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MiniMaxAI/MiniMax-M1-40k",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M2",
          "line": 30,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=30;row=29",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 82.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MiniMaxAI/MiniMax-M2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/2a/73/2a734d0f-7fcf-48d0-bea6-ad17cd749ee4.json",
          "line": 44,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 75.7,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=44;row=43",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 75.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MiniMaxAI/MiniMax-Text-01",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/33/3e/333e0d9f-7e39-4da3-96cb-0501d2380c03.json",
          "line": 83,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 49.93,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=83;row=82",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 49.93
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen1.5-110B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/09/ca/09ca61cf-1158-46e6-aeae-1d68f5dfbc34.json",
          "line": 109,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 38.02,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=109;row=108",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 38.02
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen1.5-14B-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/db/e1/dbe12346-25db-4fac-9421-5153fc4c661c.json",
          "line": 126,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 29.06,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=126;row=125",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 29.06
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen1.5-7B-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/85/72/8572a257-22bc-45c5-8891-d698ba6fc206.json",
          "line": 137,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 14.97,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=137;row=136",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 14.97
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2-0.5B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ff/fc/fffc978a-e6ae-41f2-bded-06bb776ee8b9.json",
          "line": 132,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.56,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=132;row=131",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 22.56
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2-1.5B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/18/ea/18ea6501-7d93-4ec4-bd82-86991bb5cc68.json",
          "line": 131,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.62,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=131;row=130",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 22.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2-1.5B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/37/79/3779cfdb-412d-49fa-94d4-5c7ae1af8ac6.json",
          "line": 105,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 40.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=105;row=104",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 40.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2-7B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5b/03/5b03aafd-e973-4ea4-850c-2ab97c1abb27.json",
          "line": 138,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 14.92,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=138;row=137",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 14.92
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-0.5B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e1/d5/e1d5921c-130a-44f8-ae14-baed069e565e.json",
          "line": 122,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 32.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=122;row=121",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 32.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-1.5B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/cc/7c/cc7c2263-9baf-45fc-b58b-64c15989e358.json",
          "line": 61,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 63.69,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=61;row=60",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 63.69
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-14B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/da/c5/dac51285-4928-4ac7-b052-510e4705981b.json",
          "line": 50,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.23,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=50;row=49",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 69.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-32B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ed/3a/ed3af7f3-5881-4ddd-bab1-75ecda64c98e.json",
          "line": 96,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 43.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=96;row=95",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 43.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/06/60/06600438-ea9f-4078-8eef-701c59af71a4.json",
          "line": 46,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.59,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=46;row=45",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 71.59
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-72B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0c/64/0c645bb3-1e9c-4400-a704-f99748356114.json",
          "line": 94,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=94;row=93",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 45.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-7B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9c/45/9c455b8c-1153-4c24-9de0-1d506297d805.json",
          "line": 143,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 0.6494348404255319,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=143;row=142",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 0.6494348404255319
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen2.5-VL-72B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/99/98/99989790-84de-4fd9-af37-1d9c28dd95b3.json",
          "line": 52,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 68.18,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=52;row=51",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 68.18
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-235B-A22B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/df/98/df9861f7-59cf-4b5e-8988-571481f47d14.json",
          "line": 27,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=27;row=26",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-235B-A22B-Instruct-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/74/e6/74e614fc-75a5-4f12-b077-cbf525141a9f.json",
          "line": 62,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 61.7,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=62;row=61",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 61.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-30B-A3B-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/46/a3/46a34cab-9475-40a1-ba07-2c0528f6bf5d.json",
          "line": 39,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.9,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=39;row=38",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 80.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-30B-A3B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/Qwen/Qwen3-4B-Instruct-2507",
          "line": 48,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=48;row=47",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 69.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-4B-Instruct-2507",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 21,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=21;row=20",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "RedHatAI/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/1b/16/1b16feea-1def-49ae-b3da-ee3af193a705.json",
          "line": 104,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 40.85,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=104;row=103",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 40.85
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "TIGER-Lab/MAmmoTH2-7B-Plus",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/1f/96/1f963bf0-4985-4d1f-a322-55865369de75.json",
          "line": 97,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 43.35,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=97;row=96",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 43.35
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "TIGER-Lab/MAmmoTH2-8B-Plus",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/17/d1/17d1bf78-2fbe-414d-b555-6d672f57ddf2.json",
          "line": 82,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 50.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=82;row=81",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 50.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "TIGER-Lab/MAmmoTH2-8x7B-Plus",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/67/a9/67a990d8-bf28-4d9e-94b8-a472bde1efbd.json",
          "line": 100,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 41.9,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=100;row=99",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 41.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "XiaomiMiMo/MiMo-7B-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/eb/b6/ebb62699-a109-4160-9f12-152a17ac9c5b.json",
          "line": 70,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 58.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=70;row=69",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 58.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "XiaomiMiMo/MiMo-7B-RL",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b0/a9/b0a990c3-c1a2-4a69-abed-e1c2c51d8763.json",
          "line": 114,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 36.93,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=114;row=113",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 36.93
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "abacusai/Llama-3-Smaug-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/aa/2b/aa2be737-1a21-4cc9-93f2-83316b7dc0cd.json",
          "line": 85,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 49.46,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=85;row=84",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 49.46
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ai21labs/AI21-Jamba-Large-1.5",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5d/0e/5d0edbf1-b369-43eb-b885-d5a87c818d1f.json",
          "line": 72,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 56.7,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=72;row=71",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 56.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "baidu/ERNIE-4.5-21B-A3B-Base-PT",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/bc/f1/bcf18d24-d600-4ed9-ba95-73f8d58569e1.json",
          "line": 49,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.5,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=49;row=48",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 69.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "baidu/ERNIE-4.5-300B-A47B-Base-PT",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e6/6b/e66bffdc-f3d1-4808-90a4-43f204aa883e.json",
          "line": 42,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 78.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=42;row=41",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 78.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "baidu/ERNIE-4.5-300B-A47B-PT",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/1a/c8/1ac85691-ef95-4bf5-8618-6bac456e2792.json",
          "line": 118,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 34.37,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=118;row=117",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 34.37
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-Coder-V2-Lite-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0c/0f/0c0f446a-6c2a-47c7-a120-4fafb90f92a3.json",
          "line": 101,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 41.57,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=101;row=100",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 41.57
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/deepseek-ai/DeepSeek-R1",
          "line": 18,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=18;row=17",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 84.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-R1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/a7/68/a768afc2-e0b4-4ea5-ba40-166136c33b3b.json",
          "line": 75,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 54.81,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=75;row=74",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 54.81
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-V2-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/65/c2/65c29cb3-af56-45f1-be1b-751630185a59.json",
          "line": 56,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 65.83,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=56;row=55",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 65.83
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-V2.5",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V3",
          "line": 59,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 64.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=59;row=58",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 64.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-V3",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V3-0324",
          "line": 35,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.2,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=35;row=34",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/DeepSeek-V3-0324",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/65/7f/657f147a-6450-42fe-a10d-e626f9000d29.json",
          "line": 117,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 35.3,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=117;row=116",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 35.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "deepseek-ai/deepseek-math-7b-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 24,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=24;row=23",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-MXFP4_MOE-dequant-bf16-vllm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 22,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=22;row=21",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q3_K_M-dequant-bf16-vllm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 23,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=23;row=22",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q4_K_M-dequant-bf16-vllm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/76/6f/766fb522-5c40-49f9-bcad-bd5496404129.json",
          "line": 93,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=93;row=92",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 45.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "google/gemma-2-9b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/06/7f/067fe2c1-7ba3-4f79-8344-e55e999ae537.json",
          "line": 134,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 21.72,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=134;row=133",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 21.72
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-3.0-2b-base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/d9/f9/d9f9b73a-5b6d-418b-8f26-f89b4d2b6792.json",
          "line": 123,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31.03,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=123;row=122",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 31.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-3.0-8b-base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4a/f2/4af2fd82-845a-4087-83eb-ae9b4b7561d1.json",
          "line": 139,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 12.34,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=139;row=138",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 12.34
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-3.1-1b-a400m-base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/d4/c0/d4c06d3d-8a53-44e5-9db6-ebffc99390d2.json",
          "line": 129,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 23.89,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=129;row=128",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 23.89
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-3.1-2b-base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/a9/31/a931868f-9ae1-4cd5-a2a7-8ea402101bbe.json",
          "line": 135,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 20.39,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=135;row=134",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 20.39
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-3.1-3b-a800m-base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5b/c9/5bc90039-eb1c-4772-aa34-1cfb7c2343b6.json",
          "line": 121,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 33.08,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=121;row=120",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 33.08
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-3.1-8b-base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/ibm-granite/granite-4.1-30b",
          "line": 60,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 64.09,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=60;row=59",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 64.09
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-4.1-30b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/ibm-granite/granite-4.1-3b",
          "line": 84,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 49.83,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=84;row=83",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 49.83
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-4.1-3b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/ibm-granite/granite-4.1-8b",
          "line": 73,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 55.99,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=73;row=72",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 55.99
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ibm-granite/granite-4.1-8b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4a/0a/4a0a0ab9-bb59-411e-a406-f6ac49a1e9f1.json",
          "line": 25,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.5,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=25;row=24",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/Intern-S1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/internlm/Intern-S2-Preview",
          "line": 2,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 88,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=2;row=1",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 88.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/Intern-S2-Preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/6f/ee/6fee4d03-47d0-4c2e-b449-978442fa002c.json",
          "line": 112,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=112;row=111",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 37.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/internlm2-math-plus-20b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/37/20/37203543-7ca5-4441-be5c-e4cbd12293c2.json",
          "line": 120,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 33.5,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=120;row=119",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 33.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/internlm2-math-plus-7b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/46/a7/46a7e0d0-c351-46ac-99f0-98aebcf3fef7.json",
          "line": 71,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 57.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=71;row=70",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 57.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/internlm3-8b-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/jdopensource/JoyAI-LLM-Flash",
          "line": 37,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.02,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=37;row=36",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.02
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "jdopensource/JoyAI-LLM-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/19/91/1991ed80-048e-428c-a9f7-e32c71bd65ea.json",
          "line": 29,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.7,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=29;row=28",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 82.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meituan-longcat/LongCat-Flash-Chat",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/meituan-longcat/LongCat-Flash-Lite",
          "line": 43,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 78.29,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=43;row=42",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 78.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meituan-longcat/LongCat-Flash-Lite",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/50/a3/50a33ec6-cf9b-4acc-9f45-61cd8356207c.json",
          "line": 63,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 61.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=63;row=62",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 61.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.1-405B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/3b/18/3b18649e-fd29-4193-a8c1-6d6d93d1ebd2.json",
          "line": 79,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.47,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=79;row=78",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 52.47
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.1-70B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/45/ba/45baf8ce-a6cc-412f-a134-2242a4af4c93.json",
          "line": 115,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 36.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=115;row=114",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 36.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.1-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/d5/a3/d5a3c88a-3fee-40b3-a8c7-8c4cc65baf5f.json",
          "line": 140,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 11.95,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=140;row=139",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 11.95
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.2-1B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/91/89/9189a395-7ef2-453e-84e0-259870efc2f7.json",
          "line": 133,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 22.17,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=133;row=132",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 22.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Llama-3.2-3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/77/50/77508bc1-dc52-4a84-9df1-deb62975c053.json",
          "line": 78,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.78,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=78;row=77",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 52.78
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Meta-Llama-3-70B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ac/75/ac75fe42-62e5-4ea9-a130-1eeacc8e9fb2.json",
          "line": 116,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 35.36,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=116;row=115",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 35.36
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Meta-Llama-3-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/3c/1f/3c1f9257-27aa-4da0-b772-60d62242a72b.json",
          "line": 103,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 40.98,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=103;row=102",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 40.98
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-llama/Meta-Llama-3-8B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/22/1e/221e4697-1155-4d8d-87df-9778573a3ad7.json",
          "line": 81,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 51.91,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=81;row=80",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 51.91
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/Phi-3-medium-128k-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/63/f9/63f9e7e6-d69f-4522-a81d-7280eacf730c.json",
          "line": 74,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 55.7,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=74;row=73",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 55.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/Phi-3-medium-4k-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ab/4d/ab4d706d-1629-46ac-8db3-cb6fac6de332.json",
          "line": 95,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 43.86,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=95;row=94",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 43.86
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/Phi-3-mini-128k-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/05/3a/053a7b11-0d89-4289-bfe4-ed3eca7c4a97.json",
          "line": 91,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45.66,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=91;row=90",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 45.66
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/Phi-3-mini-4k-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/03/b7/03b77a5a-1723-4cab-9d50-1a634bf8b06a.json",
          "line": 88,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 47.87,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=88;row=87",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 47.87
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/Phi-3.5-mini-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b2/6e/b26e73bc-2a30-4fa4-8223-148df3d82658.json",
          "line": 77,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.8,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=77;row=76",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 52.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/Phi-4-mini-instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/c2/37/c23724de-5d83-4794-abcb-9af3ce249886.json",
          "line": 47,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 70.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=47;row=46",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 70.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "microsoft/phi-4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/93/21/9321cec5-309d-4328-aa2c-709af17cff9c.json",
          "line": 125,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 30.43,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=125;row=124",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 30.43
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistral-community/Mistral-7B-v0.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b7/6c/b76c036b-b67b-4dd4-9c7f-d1938b83f254.json",
          "line": 99,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 42,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=99;row=98",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 42.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mathstral-7B-v0.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b0/f5/b0f59e78-883f-4c57-b728-b983c859f23f.json",
          "line": 110,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.93,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=110;row=109",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 37.93
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Ministral-8B-Instruct-2410",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9d/8d/9d8dc94b-067c-478d-aa99-874b71721fc3.json",
          "line": 124,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 30.84,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=124;row=123",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 30.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-7B-Instruct-v0.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b7/99/b799f854-5a9c-413c-b766-ab127e0cd103.json",
          "line": 55,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 65.91,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=55;row=54",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 65.91
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Large-Instruct-2407",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/03/98/03985ef7-288d-4421-9257-ce6b77dcd858.json",
          "line": 53,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 67.94,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=53;row=52",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 67.94
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Large-Instruct-2411",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/be/91/be919504-16fb-4669-8d9b-3bfdfd2e6364.json",
          "line": 106,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 39.77,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=106;row=105",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 39.77
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Nemo-Base-2407",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/d4/89/d489ea6c-1287-43bd-aa64-5193c09331b2.json",
          "line": 76,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 54.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=76;row=75",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 54.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Small-24B-Base-2501",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/71/31/7131c1f0-8fce-4050-8117-51d9b5e651ca.json",
          "line": 86,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 48.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=86;row=85",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 48.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Small-Instruct-2409",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/2b/dd/2bddced4-34a8-4a1c-b582-90bfe03346f1.json",
          "line": 102,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 41.03,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=102;row=101",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 41.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mixtral-8x7B-v0.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4a/ae/4aae2732-e24b-42a2-9dc1-a35cd215a8ec.json",
          "line": 38,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=38;row=37",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/2a/e5/2ae5bb55-f5e3-4aa2-beef-141f450a4562.json",
          "line": 58,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 65.1,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=58;row=57",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 65.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-Base-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 20,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 83.73,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=20;row=19",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 83.73
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
          "line": 6,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.8,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=6;row=5",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 86.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
          "line": 7,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.8,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=7;row=6",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 86.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
          "line": 31,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.94,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=31;row=30",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.94
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "line": 32,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.62,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=32;row=31",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/8b/b3/8bb31817-e160-4285-a781-272851ae2c41.json",
          "line": 64,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.5,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=64;row=63",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 60.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/Nemotron-H-56B-Base-8K",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e4/8d/e48dbb1a-9e1a-4c99-abbf-cb32942b48e7.json",
          "line": 40,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.8,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=40;row=39",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 80.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-120b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ca/ef/caef6dbc-0f07-4491-9009-2b0d26dc9724.json",
          "line": 45,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 73.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=45;row=44",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 73.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-20b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/skt/A.X-K1",
          "line": 33,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/A.X-K1.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.5,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=33;row=32",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "skt/A.X-K1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://arxiv.org/abs/2602.10604",
          "line": 17,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=17;row=16",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 84.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4b/ff/4bffdd42-60c3-41a9-9ef8-e5849e114d9e.json",
          "line": 54,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 67.3,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=54;row=53",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 67.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hunyuan-A13B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5c/9f/5c9f18ed-3247-4af2-b44f-2a504735554b.json",
          "line": 66,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 60.2,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=66;row=65",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 60.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Tencent-Hunyuan-Large",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "line": 11,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86.2,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=11;row=10",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 86.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "upstage/Solar-Open2-250B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/04/c7/04c77f44-dff1-4cc7-80a1-35f4e9e9237f.json",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 84.6,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 84.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.5",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/07/43/074397da-f5e5-441c-bc45-ac16b599ca7c.json",
          "line": 34,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 81.4,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=34;row=33",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 81.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.5-Air",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "mmlu-pro"
      ],
      "examples": [
        {
          "artifact": "hf-mmlu-pro/candidates.jsonl",
          "benchmarkId": "mmlu-pro",
          "benchmarkName": "MMLU-Pro",
          "evidenceUrl": "https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/92/ff/92ff34d6-a9d7-48af-a31e-a72c2367edff.json",
          "line": 87,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/mmlu_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 47.92,
          "sourceId": "hf-mmlu-pro",
          "sourceLabel": "Hugging Face · MMLU-Pro leaderboard API",
          "sourceLocator": "rank=87;row=86",
          "sourceUrl": "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard",
          "unit": "percent",
          "value": 47.92
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/glm-4-9b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-mmlu-pro",
      "sourceIds": [
        "hf-mmlu-pro"
      ],
      "sourceLabels": [
        "Hugging Face · MMLU-Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/TIGER-Lab/MMLU-Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/CohereLabs/North-Mini-Code-1.0",
          "line": 36,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench-pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 40.2,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=36;row=35",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 40.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/North-Mini-Code-1.0",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 12,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 58.6,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=12;row=11",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 58.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS.2",
          "line": 33,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 44.5,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=33;row=32",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 44.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MuVeraAI/Laguna-XS.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://scale.com/leaderboard/swe_bench_pro_public",
          "line": 40,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 21.41,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=40;row=39",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 21.41
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Qwen/Qwen3-235B-A22B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/dots-studio/dots3-note-prev",
          "line": 7,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/dots3-note-prev.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 61,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=7;row=6",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 61.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "dots-studio/dots3-note-prev",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-3.0-flash",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 56.6,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 56.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-3.0-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
          "line": 25,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Muse-Glimmer-30B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 51.2,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=25;row=24",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 51.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-models/Muse-Glimmer-30B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://scale.com/leaderboard/swe_bench_pro_public",
          "line": 39,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 27.67,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=39;row=38",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 27.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://scale.com/leaderboard/swe_bench_pro_public",
          "line": 41,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 16.2,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=41;row=40",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 16.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "openai/gpt-oss-120b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B",
          "line": 27,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Ornith-1.0-35B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 50.4,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=27;row=26",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 50.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.0-35B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B",
          "line": 4,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Ornith-1.0-397B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 62.2,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=4;row=3",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 62.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.0-397B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B",
          "line": 35,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Ornith-1.0-9B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 42.9,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=35;row=34",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 42.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.0-9B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-35B-A3B",
          "line": 8,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-35b-a3b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.6,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=8;row=7",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 59.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-35B-A3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-397B",
          "line": 2,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-397b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 65.1,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=2;row=1",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 65.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-397B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-9B",
          "line": 31,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-9b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 47.5,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=31;row=30",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 47.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-9B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-M.1",
          "line": 29,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 49.2,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=29;row=28",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 49.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-M.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-S-2.1",
          "line": 9,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 59.4,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=9;row=8",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 59.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-S-2.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS-2.1",
          "line": 30,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 47.6,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=30;row=29",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 47.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-XS-2.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS.2",
          "line": 32,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 46.3,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=32;row=31",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 46.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-XS.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/stepfun-ai/Step-3.7-Flash",
          "line": 17,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 56.3,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=17;row=16",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 56.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.7-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3",
          "line": 14,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 57.9,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=14;row=13",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 57.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling",
          "line": 23,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 54.3,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=23;row=22",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 54.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling-Small",
          "line": 20,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 55.9,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=20;row=19",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 55.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling-Small",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-pro"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-pro/candidates.jsonl",
          "benchmarkId": "swebench-pro",
          "benchmarkName": "SWE-bench Pro",
          "evidenceUrl": "https://scale.com/leaderboard/swe_bench_pro_public",
          "line": 42,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_pro.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 9.67,
          "sourceId": "hf-swebench-pro",
          "sourceLabel": "Hugging Face · SWE-bench Pro leaderboard API",
          "sourceLocator": "rank=42;row=41",
          "sourceUrl": "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard",
          "unit": "percent",
          "value": 9.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.6",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-pro",
      "sourceIds": [
        "hf-swebench-pro"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Pro leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/ScaleAI/SWE-bench_Pro/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/CohereLabs/North-Mini-Code-1.0",
          "line": 48,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench-verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 67.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=48;row=47",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 67.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/North-Mini-Code-1.0",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 9,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=9;row=8",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 80.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M2",
          "line": 45,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=45;row=44",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 69.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MiniMaxAI/MiniMax-M2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS.2",
          "line": 47,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 68.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=47;row=46",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 68.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MuVeraAI/Laguna-XS.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/papers/2603.16790",
          "line": 24,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74.8,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=24;row=23",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 74.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "Multilingual-Multimodal-NLP/IndustrialCoder",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/dots-studio/dots3-note-prev",
          "line": 13,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/dots3-note-prev.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 78.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=13;row=12",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 78.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "dots-studio/dots3-note-prev",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/papers/2510.02387",
          "line": 66,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 53.9,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=66;row=65",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 53.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "facebook/cwm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-2.6-1T",
          "line": 34,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 72.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=34;row=33",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 72.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-2.6-1T",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ling-2.6-flash",
          "line": 52,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 61.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=52;row=51",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 61.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ling-2.6-flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/inclusionAI/Ring-2.6-1T",
          "line": 29,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=29;row=28",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 74.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "inclusionAI/Ring-2.6-1T",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/internlm/Intern-S2-Preview",
          "line": 49,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 64,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=49;row=48",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 64.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "internlm/Intern-S2-Preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
          "line": 20,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Muse-Glimmer-30B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 76,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=20;row=19",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 76.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "meta-models/Muse-Glimmer-30B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/mindlab-research/Macaron-V1-Coding-Venti/blob/0a7ec8fdf72e0757decabd90cc3305bed1fed1c6/README.md#L68-L76",
          "line": 3,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 85.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=3;row=2",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 85.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mindlab-research/Macaron-V1-Coding-Venti",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/mindlab-research/Macaron-V1-Tall/blob/d0b2199c3572d336bcfe9e027ec519314cf608bd/assets/v1_benchmark.png",
          "line": 23,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 75.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=23;row=22",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 75.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mindlab-research/Macaron-V1-Tall",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/mindlab-research/Macaron-V1-Venti",
          "line": 2,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 85.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=2;row=1",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 85.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mindlab-research/Macaron-V1-Venti",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 77.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 77.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mistralai/Mistral-Medium-3.5-128B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2-Thinking",
          "line": 37,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.3,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=37;row=36",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 71.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Thinking",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
          "line": 36,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 71.9,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=36;row=35",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 71.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
          "line": 44,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.7,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=44;row=43",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 69.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
          "line": 75,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 51.56,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=75;row=74",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 51.56
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "line": 73,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 52.8,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=73;row=72",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 52.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B",
          "line": 22,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Ornith-1.0-35B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 75.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=22;row=21",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 75.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.0-35B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B",
          "line": 4,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Ornith-1.0-397B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 82.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=4;row=3",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 82.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.0-397B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B",
          "line": 46,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/Ornith-1.0-9B.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=46;row=45",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 69.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.0-9B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-35B-A3B",
          "line": 11,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-35b-a3b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 79,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=11;row=10",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 79.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-35B-A3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-397B",
          "line": 1,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-397b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 86,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=1;row=0",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 86.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-397B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/ornith-ai/Ornith-1.5-9B",
          "line": 41,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/ornith-1.5-9b.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 70.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=41;row=40",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 70.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "ornith-ai/Ornith-1.5-9B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-M.1",
          "line": 25,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=25;row=24",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 74.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-M.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS-2.1",
          "line": 38,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 70.9,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=38;row=37",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 70.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-XS-2.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS.2",
          "line": 43,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 69.9,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=43;row=42",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 69.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-XS.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://arxiv.org/abs/2602.10604",
          "line": 26,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=26;row=25",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 74.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3",
          "line": 14,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 78,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=14;row=13",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 78.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3-preview",
          "line": 27,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe_bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 74.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=27;row=26",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 74.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3-preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling",
          "line": 17,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 77.6,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=17;row=16",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 77.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/thinkingmachines/Inkling-Small",
          "line": 8,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 80.2,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=8;row=7",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 80.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "thinkingmachines/Inkling-Small",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "hf-swebench-verified/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "line": 42,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/swe-bench_verified.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 70.4,
          "sourceId": "hf-swebench-verified",
          "sourceLabel": "Hugging Face · SWE-bench Verified leaderboard API",
          "sourceLocator": "rank=42;row=41",
          "sourceUrl": "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard",
          "unit": "percent",
          "value": 70.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "upstage/Solar-Open2-250B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-swebench-verified",
      "sourceIds": [
        "hf-swebench-verified"
      ],
      "sourceLabels": [
        "Hugging Face · SWE-bench Verified leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/SWE-bench/SWE-bench_Verified/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/CohereLabs/North-Mini-Code-1.0",
          "line": 23,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench-v2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 36,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=23;row=22",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 36.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "CohereLabs/North-Mini-Code-1.0",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "line": 6,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench_2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 66.7,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=6;row=5",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 66.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "INCModel/Kimi-K2.6-MXFP4-CT-AutoRound",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 33,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 30,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=33;row=32",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 30.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MiniMaxAI/MiniMax-M2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS.2",
          "line": 32,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench-2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 30.1,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=32;row=31",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 30.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "MuVeraAI/Laguna-XS.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 28,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench_2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=28;row=27",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 31.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "RedHatAI/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 31,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench_2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=31;row=30",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 31.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-MXFP4_MOE-dequant-bf16-vllm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 29,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench_2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=29;row=28",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 31.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q3_K_M-dequant-bf16-vllm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 30,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench_2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=30;row=29",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 31.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "exolabs/NVIDIA-Nemotron-3-Super-120B-A12B-UD-Q4_K_M-dequant-bf16-vllm",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/mindlab-research/Macaron-V1-Venti/blob/main/README.md#evaluation",
          "line": 1,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench_2_1.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 87.6,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=1;row=0",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 87.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "mindlab-research/Macaron-V1-Venti",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 35,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 27.8,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=35;row=34",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 27.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 24,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 35.7,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=24;row=23",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 35.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "moonshotai/Kimi-K2-Thinking",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "line": 27,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench_2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 31,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=27;row=26",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 31.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://arxiv.org/pdf/2602.21193",
          "line": 39,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench_2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 20.2,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=39;row=38",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 20.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/Nemotron-Terminal-14B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://arxiv.org/pdf/2602.21193",
          "line": 36,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench_2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 27.4,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=36;row=35",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 27.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/Nemotron-Terminal-32B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://arxiv.org/pdf/2602.21193",
          "line": 40,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench_2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 13,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=40;row=39",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 13.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "nvidia/Nemotron-Terminal-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-M.1",
          "line": 18,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench-2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 45.8,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=18;row=17",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 45.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-M.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS-2.1",
          "line": 21,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench-2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 37.5,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=21;row=20",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 37.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-XS-2.1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/poolside/Laguna-XS.2",
          "line": 25,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal-bench-2.0.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 35.7,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=25;row=24",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 35.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "poolside/Laguna-XS.2",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://arxiv.org/abs/2602.10604",
          "line": 16,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench_2.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 51,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=16;row=15",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 51.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "stepfun-ai/Step-3.5-Flash",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://huggingface.co/tencent/Hy3-preview",
          "line": 12,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 54.4,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=12;row=11",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 54.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "tencent/Hy3-preview",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "terminal-bench"
      ],
      "examples": [
        {
          "artifact": "hf-terminal-bench/candidates.jsonl",
          "benchmarkId": "terminal-bench",
          "benchmarkName": "Terminal-Bench 2.1",
          "evidenceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
          "line": 37,
          "metricId": "score",
          "observedAt": null,
          "protocol": {
            "filename": ".eval_results/terminal_bench.yaml",
            "harness": "huggingface-eval-results",
            "verified": false
          },
          "rawValue": 24.5,
          "sourceId": "hf-terminal-bench",
          "sourceLabel": "Hugging Face · Terminal-Bench leaderboard API",
          "sourceLocator": "rank=37;row=36",
          "sourceUrl": "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard",
          "unit": "percent",
          "value": 24.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "score"
      ],
      "modelRef": "zai-org/GLM-4.6",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "hf-terminal-bench",
      "sourceIds": [
        "hf-terminal-bench"
      ],
      "sourceLabels": [
        "Hugging Face · Terminal-Bench leaderboard API"
      ],
      "sourceUrls": [
        "https://huggingface.co/api/datasets/harborframework/terminal-bench-2.0/leaderboard"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2606,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.10316952943197616,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=47",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.10316952943197616
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Gemini 3 Flash",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2610,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.20609789716978127,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=51",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.20609789716978127
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Gemma 4 31B",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2609,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.17609393328487202,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=50",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.17609393328487202
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Grok 4.3",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2604,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.09856786775357175,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=45",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.09856786775357175
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Grok 4.3 (High)",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2603,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.09440410862150922,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=44",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.09440410862150922
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Grok Build 0.1",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2589,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.004218449993691898,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=30",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.004218449993691898
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Hy3",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2601,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.06740904363796141,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=42",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.06740904363796141
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Inkling",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2599,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.052211887180652866,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=40",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.052211887180652866
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Mistral Medium 3.5",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2587,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.000977217234170596,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=28",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.000977217234170596
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Muse Spark 1.1",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2584,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 0.01603087510299009,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=25",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": 0.01603087510299009
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Muse Spark 1.2 (xHigh)",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2608,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.12669786915868378,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=49",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.12669786915868378
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Nemotron 3 Ultra",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-agent"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-agent",
          "benchmarkName": "Agent Arena",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2602,
          "metricId": "ips",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "agent",
            "category": "overall",
            "harness": "arena-agent",
            "rating_method": "ips",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": -0.08786498735008978,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=agent;split=latest;row_idx=43",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "fraction",
          "value": -0.08786498735008978
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ips"
      ],
      "modelRef": "Solar Pro 4",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2520,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1006.180528118307,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=33",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1006.180528118307
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "api-gpt-4o-search",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 387,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 938.7535456861812,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=386",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 938.7535456861812
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "chatglm2-6b",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2511,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1148.2792594379139,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=24",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1148.2792594379139
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "claude-opus-4-1-search",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 365,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1057.3697134369597,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=364",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1057.3697134369597
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "codellama-70b-instruct",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2519,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1022.5527036488551,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=32",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1022.5527036488551
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "diffbot-small-xl",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 355,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1081.101314832244,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=354",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1081.101314832244
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "dolphin-2.2.1-mistral-7b",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 368,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1054.4810894868247,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=367",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1054.4810894868247
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "falcon-180b-chat",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2513,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1141.6494107389728,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=26",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1141.6494107389728
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-2.5-pro-grounding",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2501,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1198.0900913160758,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=14",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1198.0900913160758
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3-flash-grounding",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2496,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1207.3489389354843,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=9",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1207.3489389354843
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3-pro-grounding",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2495,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1210.4636662372704,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=8",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1210.4636662372704
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemini-3.1-pro-grounding",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 201,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1334.2283079805243,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=200",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1334.2283079805243
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-3-12b-it",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 235,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1290.8175201707122,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=234",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1290.8175201707122
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gemma-3-4b-it",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 384,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 956.3073115480572,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=383",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 956.3073115480572
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "gpt4all-13b-snoozy",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2507,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1170.9411924302385,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=20",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1170.9411924302385
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "grok-4-1-fast-search",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 370,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1053.5060173821591,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=369",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1053.5060173821591
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "guanaco-33b",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 394,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 833.6353946828292,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=393",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 833.6353946828292
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama-13b",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 345,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1098.2192428579874,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=344",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1098.2192428579874
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "llama2-70b-steerlm-chat",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 361,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1068.73682076947,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=360",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1068.73682076947
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "mpt-30b-chat",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 337,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1112.1679175623235,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=336",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1112.1679175623235
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "nous-hermes-2-mixtral-8x7b-dpo",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2512,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1144.2640168648277,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=25",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1144.2640168648277
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "o3-search",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2517,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1130.2067048282445,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=30",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1130.2067048282445
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ppl-sonar-pro-high",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-search"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-search",
          "benchmarkName": "Arena Search",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 2515,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-24",
          "protocol": {
            "arena_config": "search",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "system"
          },
          "rawValue": 1138.6014177151164,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=search;split=latest;row_idx=28",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1138.6014177151164
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "ppl-sonar-reasoning-pro-high",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-24",
      "observedAtMin": "2026-08-24",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 373,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1042.150044481211,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=372",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1042.150044481211
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "smollm2-1.7b-instruct",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 354,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1083.2157336108166,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=353",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1083.2157336108166
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "solar-10.7b-instruct-v1.0",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 392,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 866.6988480586815,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=391",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 866.6988480586815
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "stablelm-tuned-alpha-7b",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 374,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1038.5671870772248,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=373",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1038.5671870772248
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "stripedhyena-nous-7b",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "arena-text"
      ],
      "examples": [
        {
          "artifact": "lmarena-hf-dataset/candidates.jsonl",
          "benchmarkId": "arena-text",
          "benchmarkName": "Arena Text",
          "evidenceUrl": "https://datasets-server.huggingface.co/rows",
          "line": 363,
          "metricId": "arena_score_bt",
          "observedAt": "2026-08-21",
          "protocol": {
            "arena_config": "text",
            "category": "overall",
            "harness": "arena-human-preference",
            "rating_method": "bradley-terry",
            "split": "latest",
            "subject_type": "model"
          },
          "rawValue": 1058.497023185306,
          "sourceId": "lmarena-hf-dataset",
          "sourceLabel": "Arena · official Hugging Face leaderboard dataset",
          "sourceLocator": "dataset=lmarena-ai/leaderboard-dataset;config=text;split=latest;row_idx=362",
          "sourceUrl": "https://datasets-server.huggingface.co/rows",
          "unit": "rating",
          "value": 1058.497023185306
        }
      ],
      "latestRetrievedAt": "2026-08-27T06:23:04Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "arena_score_bt"
      ],
      "modelRef": "zephyr-7b-alpha",
      "numericRowCount": 1,
      "observedAtMax": "2026-08-21",
      "observedAtMin": "2026-08-21",
      "rowCount": 1,
      "sourceId": "lmarena-hf-dataset",
      "sourceIds": [
        "lmarena-hf-dataset"
      ],
      "sourceLabels": [
        "Arena · official Hugging Face leaderboard dataset"
      ],
      "sourceUrls": [
        "https://datasets-server.huggingface.co/rows"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 16,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-13",
          "protocol": {
            "command": "aider --model mistral/codestral-latest",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=391;dirname=2025-01-13-18-17-25--codestral-whole2",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Codestral 25.01",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-13",
      "observedAtMin": "2025-01-13",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 7,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-21",
          "protocol": {
            "command": "aider --model deepseek/deepseek-chat",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 5.3,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=157;dirname=2024-12-21-20-56-21--polyglot-deepseek-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 5.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "DeepSeek Chat V2.5",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-21",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 14,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-25",
          "protocol": {
            "command": "aider --model deepseek/deepseek-chat",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 22.7,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=339;dirname=2024-12-25-13-31-51--deepseekv3preview-diff2",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 22.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "DeepSeek Chat V3 (prev)",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-25",
      "observedAtMin": "2024-12-25",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 17,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-20",
          "protocol": {
            "command": "aider --model deepseek/deepseek-reasoner",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 26.7,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=417;dirname=2025-01-20-19-11-38--ds-turns-upd-cur-msgs-fix-with-summarizer",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 26.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "DeepSeek R1",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-20",
      "observedAtMin": "2025-01-20",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 18,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-23",
          "protocol": {
            "command": "aider --architect --model r1 --editor-model sonnet",
            "edit_format": "architect",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 27.1,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=443;dirname=2025-01-23-19-14-48--r1-architect-sonnet",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 27.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "DeepSeek R1 + claude-3-5-sonnet-20241022",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-23",
      "observedAtMin": "2025-01-23",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 31,
          "metricId": "pass_rate_1",
          "observedAt": "2025-03-24",
          "protocol": {
            "command": "aider --model deepseek/deepseek-chat",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 28.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=784;dirname=2025-03-24-15-41-33--deepseek-v3-0324-polyglot-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 28.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "DeepSeek V3 (0324)",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-24",
      "observedAtMin": "2025-03-24",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 1,
          "metricId": "pass_rate_1",
          "observedAt": "2025-02-25",
          "protocol": {
            "command": "aider --model gemini/gemini-2.0-pro-exp-02-05",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 20.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1;dirname=2025-02-25-20-23-07--gemini-pro",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 20.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Gemini 2.0 Pro exp-02-05",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-25",
      "observedAtMin": "2025-02-25",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 32,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-12",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-pro-preview-03-25",
            "edit_format": "diff-fenced",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 40.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=810;dirname=2025-04-12-04-55-50--gemini-25-pro-diff-fenced",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 40.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Gemini 2.5 Pro Preview 03-25",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-12",
      "observedAtMin": "2025-04-12",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 46,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-07",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-pro-preview-05-06",
            "edit_format": "diff-fenced",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 36.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1174;dirname=2025-05-07-19-32-40--gemini0506-diff-fenced-completion_cost",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 36.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Gemini 2.5 Pro Preview 05-06",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-07",
      "observedAtMin": "2025-05-07",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 36,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-10",
          "protocol": {
            "command": "aider --model openrouter/x-ai/grok-3-beta",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 22.2,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=914;dirname=2025-04-10-04-21-31--grok3-diff-exuser",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 22.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Grok 3 Beta",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-10",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 38,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-10",
          "protocol": {
            "command": "aider --model xai/grok-3-mini-beta --reasoning-effort high",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 17.3,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=966;dirname=2025-04-10-23-59-02--xai-grok3-mini-whole-high",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 17.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Grok 3 Mini Beta (high)",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-10",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 37,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-10",
          "protocol": {
            "command": "aider --model openrouter/x-ai/grok-3-mini-beta",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 11.1,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=940;dirname=2025-04-10-18-47-24--grok3-mini-whole-exuser",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 11.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Grok 3 Mini Beta (low)",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-10",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 63,
          "metricId": "pass_rate_1",
          "observedAt": "2025-07-17",
          "protocol": {
            "command": "aider --model openrouter/moonshotai/kimi-k2",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 20.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1659;dirname=2025-07-17-17-41-54--kimi-k2-diff-or-pricing",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 20.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Kimi K2",
      "numericRowCount": 1,
      "observedAtMax": "2025-07-17",
      "observedAtMin": "2025-07-17",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 35,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-06",
          "protocol": {
            "command": "aider --model nvidia_nim/meta/llama-4-maverick-17b-128e-instruct",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=888;dirname=2025-04-06-08-39-52--llama-4-maverick-17b-128e-instruct-polyglot-whole",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Llama 4 Maverick",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-06",
      "observedAtMin": "2025-04-06",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 39,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-10",
          "protocol": {
            "command": "aider --model openrouter/openrouter/optimus-alpha",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 21.3,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=992;dirname=2025-04-10-19-02-44--oalpha-diff-exsys",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 21.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Optimus Alpha",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-10",
      "observedAtMin": "2025-04-10",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 34,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-04",
          "protocol": {
            "command": "aider --model openrouter/openrouter/quasar-alpha",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 21.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=862;dirname=2025-04-04-02-57-25--qalpha-diff-exsys",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 21.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Quasar Alpha",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-04",
      "observedAtMin": "2025-04-04",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 27,
          "metricId": "pass_rate_1",
          "observedAt": "2025-03-06",
          "protocol": {
            "command": "aider --model fireworks_ai/accounts/fireworks/models/qwq-32b",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 8.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=678;dirname=2025-03-06-17-40-24--qwq32b-diff-temp-topp-ex-sys-remind-user-for-real",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 8.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "QwQ-32B",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-06",
      "observedAtMin": "2025-03-06",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 28,
          "metricId": "pass_rate_1",
          "observedAt": "2025-03-07",
          "protocol": {
            "command": "aider --model fireworks_ai/accounts/fireworks/models/qwq-32b --architect",
            "edit_format": "architect",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 9.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=704;dirname=2025-03-07-15-11-27--qwq32b-arch-temp-topp-again",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 9.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "QwQ-32B + Qwen 2.5 Coder Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-07",
      "observedAtMin": "2025-03-07",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 48,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-09",
          "protocol": {
            "command": "aider --model openai/qwen3-235b-a22b",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 28.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1228;dirname=2025-05-09-17-02-02--qwen3-235b-a22b.unthink_16k_diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 28.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Qwen3 235B A22B diff, no think, Alibaba API",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-09",
      "observedAtMin": "2025-05-09",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 47,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-08",
          "protocol": {
            "command": "aider --model openrouter/qwen/qwen3-32b",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 14.2,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1200;dirname=2025-05-08-03-20-24--qwen3-32b-default",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 14.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "Qwen3 32B",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-08",
      "observedAtMin": "2025-05-08",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 23,
          "metricId": "pass_rate_1",
          "observedAt": "2025-02-15",
          "protocol": {
            "command": "aider --model chatgpt-4o-latest",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 223
          },
          "rawValue": 9.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=574;dirname=2025-02-15-19-51-22--chatgpt4o-feb15-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 9.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "chatgpt-4o-latest (2025-02-15)",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-15",
      "observedAtMin": "2025-02-15",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 33,
          "metricId": "pass_rate_1",
          "observedAt": "2025-03-29",
          "protocol": {
            "command": "aider --model chatgpt-4o-latest",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 16.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=836;dirname=2025-03-29-05-24-55--chatgpt4o-mar28-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 16.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "chatgpt-4o-latest (2025-03-29)",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-29",
      "observedAtMin": "2025-03-29",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 8,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-21",
          "protocol": {
            "command": "aider --model claude-3-5-haiku-20241022",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 7.1,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=183;dirname=2024-12-21-21-46-27--polyglot-haiku-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 7.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "claude-3-5-haiku-20241022",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-21",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 3,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-17",
          "protocol": {
            "command": "aider --model claude-3-5-sonnet-20241022",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 22.2,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=53;dirname=2025-01-17-19-44-33--sonnet-baseline-jan-17",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 22.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "claude-3-5-sonnet-20241022",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-17",
      "observedAtMin": "2025-01-17",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 25,
          "metricId": "pass_rate_1",
          "observedAt": "2025-02-24",
          "protocol": {
            "command": "aider --model anthropic/claude-3-7-sonnet-20250219 --thinking-tokens 32k",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 29.3,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=626;dirname=2025-02-24-21-47-23--sonnet37-diff-think-32k-64k",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 29.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "claude-3-7-sonnet-20250219 (32k thinking tokens)",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 24,
          "metricId": "pass_rate_1",
          "observedAt": "2025-02-24",
          "protocol": {
            "command": "aider --model sonnet",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 24.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=600;dirname=2025-02-24-19-54-07--sonnet37-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 24.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "claude-3-7-sonnet-20250219 (no thinking)",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-24",
      "observedAtMin": "2025-02-24",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 50,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-24",
          "protocol": {
            "command": "aider --model claude-sonnet-4-20250514",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 25.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1284;dirname=2025-05-24-22-10-36--sonnet4-diff-exuser-think32k",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 25.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "claude-sonnet-4-20250514 (32k thinking)",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-24",
      "observedAtMin": "2025-05-24",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 49,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-24",
          "protocol": {
            "command": "aider --model claude-sonnet-4-20250514",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 20.4,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1256;dirname=2025-05-24-21-17-54--sonnet4-diff-exuser",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 20.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "claude-sonnet-4-20250514 (no thinking)",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-24",
      "observedAtMin": "2025-05-24",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 29,
          "metricId": "pass_rate_1",
          "observedAt": "2025-03-14",
          "protocol": {
            "command": "OPENAI_API_BASE=https://api.cohere.ai/compatibility/v1 aider --model openai/command-a-03-2025-quality",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 2.2,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=732;dirname=2025-03-14-23-40-00--cmda-quality-whole2",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 2.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "command-a-03-2025-quality",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-14",
      "observedAtMin": "2025-03-14",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 12,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-22",
          "protocol": {
            "command": "aider --model gemini/gemini-2.0-flash-exp",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 11.6,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=287;dirname=2024-12-22-20-08-13--gemini-2.0-flash-exp-polyglot-whole",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 11.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.0-flash-exp",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-22",
      "observedAtMin": "2024-12-22",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 22,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-21",
          "protocol": {
            "command": "aider --model gemini/gemini-2.0-flash-thinking-exp-01-21",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 5.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=548;dirname=2025-01-21-22-51-49--gemini-2.0-flash-thinking-exp-01-21-polyglot-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 5.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.0-flash-thinking-exp-01-21",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-21",
      "observedAtMin": "2025-01-21",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 45,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-20",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-flash-preview-04-17",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 21.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1148;dirname=2025-04-20-19-54-31--flash25-diff-no-think",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 21.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.5-flash-preview-04-17 (default)",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-20",
      "observedAtMin": "2025-04-20",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 54,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-25",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-flash-preview-05-20",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 26.2,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1399;dirname=2025-05-25-22-58-44--flash25-05-20-24k-think",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 26.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20 (24k think)",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-25",
      "observedAtMin": "2025-05-25",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 53,
          "metricId": "pass_rate_1",
          "observedAt": "2025-05-26",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-flash-preview-05-20",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 20.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1370;dirname=2025-05-26-15-56-31--flash25-05-20-24k-think # dirname is misleading",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 20.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.5-flash-preview-05-20 (no think)",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-26",
      "observedAtMin": "2025-05-26",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 56,
          "metricId": "pass_rate_1",
          "observedAt": "2025-06-06",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-pro-preview-06-05 --thinking-tokens 32k",
            "edit_format": "diff-fenced",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 46.2,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1456;dirname=2025-06-06-16-36-21--gemini0605-32k-think-diff-fenced",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 46.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.5-pro-preview-06-05 (32k think)",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-06",
      "observedAtMin": "2025-06-06",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 55,
          "metricId": "pass_rate_1",
          "observedAt": "2025-06-06",
          "protocol": {
            "command": "aider --model gemini/gemini-2.5-pro-preview-06-05",
            "edit_format": "diff-fenced",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 44.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1428;dirname=2025-06-06-18-38-56--gemini0605-diff-fenced",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 44.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-2.5-pro-preview-06-05 (default think)",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-06",
      "observedAtMin": "2025-06-06",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 11,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-22",
          "protocol": {
            "command": "aider --model gemini/gemini-exp-1206",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 19.6,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=261;dirname=2024-12-22-18-43-25--gemini-exp-1206-polyglot-whole-2",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 19.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemini-exp-1206",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-22",
      "observedAtMin": "2024-12-22",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 30,
          "metricId": "pass_rate_1",
          "observedAt": "2025-03-15",
          "protocol": {
            "command": "aider --model openrouter/google/gemma-3-27b-it",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 1.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=758;dirname=2025-03-15-01-21-24--gemma3-27b-or",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 1.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gemma-3-27b-it",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-15",
      "observedAtMin": "2025-03-15",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 40,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-14",
          "protocol": {
            "command": "aider --model gpt-4.1",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 20.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1018;dirname=2025-04-14-21-05-54--gpt41-diff-exuser",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 20.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4.1",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-14",
      "observedAtMin": "2025-04-14",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 41,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-14",
          "protocol": {
            "command": "aider --model gpt-4.1-mini",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 11.1,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1044;dirname=2025-04-14-21-27-53--gpt41mini-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 11.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4.1-mini",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-14",
      "observedAtMin": "2025-04-14",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 42,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-14",
          "protocol": {
            "command": "aider --model gpt-4.1-nano",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 3.1,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1070;dirname=2025-04-14-22-46-01--gpt41nano-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 3.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4.1-nano",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-14",
      "observedAtMin": "2025-04-14",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 26,
          "metricId": "pass_rate_1",
          "observedAt": "2025-02-27",
          "protocol": {
            "command": "aider --model openai/gpt-4.5-preview",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 224
          },
          "rawValue": 22.3,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=652;dirname=2025-02-27-20-26-15--gpt45-diff3",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 22.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4.5-preview",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-27",
      "observedAtMin": "2025-02-27",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 5,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-30",
          "protocol": {
            "command": "aider --model gpt-4o-2024-08-06",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=105;dirname=2024-12-30-20-44-54--gpt4o-ex-as-sys-clean-prompt",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4o-2024-08-06",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-30",
      "observedAtMin": "2024-12-30",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 4,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-30",
          "protocol": {
            "command": "aider --model gpt-4o-2024-11-20",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=79;dirname=2024-12-30-20-57-12--gpt-4o-2024-11-20-ex-as-sys",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4o-2024-11-20",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-30",
      "observedAtMin": "2024-12-30",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 2,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-21",
          "protocol": {
            "command": "aider --model gpt-4o-mini-2024-07-18",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 0.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=27;dirname=2024-12-21-18-41-18--polyglot-gpt-4o-mini",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 0.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-4o-mini-2024-07-18",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-21",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 64,
          "metricId": "pass_rate_1",
          "observedAt": "2025-08-06",
          "protocol": {
            "command": "aider --model openrouter/openai/gpt-oss-120b --reasoning-effort high",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 13.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1687;dirname=2025-08-06-04-54-48--gpt-oss-120b-high-polyglot",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 13.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "gpt-oss-120b (high)",
      "numericRowCount": 1,
      "observedAtMax": "2025-08-06",
      "observedAtMin": "2025-08-06",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 6,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-21",
          "protocol": {
            "command": "aider --model openrouter/openai/o1",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 224
          },
          "rawValue": 23.7,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=131;dirname=2024-12-21-19-23-03--polyglot-o1-hard-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 23.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o1-2024-12-17 (high)",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-21",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 10,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-22",
          "protocol": {
            "command": "aider --model o1-mini",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 5.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=235;dirname=2024-12-22-21-26-35--polyglot-o1mini-whole",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 5.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o1-mini-2024-09-12",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-22",
      "observedAtMin": "2024-12-22",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 59,
          "metricId": "pass_rate_1",
          "observedAt": "2025-06-25",
          "protocol": {
            "command": "aider --model o3",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 40.9,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1542;dirname=2025-06-25-20-30-16--o3-price-reduction",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 40.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o3",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-25",
      "observedAtMin": "2025-06-25",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 58,
          "metricId": "pass_rate_1",
          "observedAt": "2025-06-25",
          "protocol": {
            "command": "aider --model o3 --reasoning-effort high",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 40.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1513;dirname=2025-06-25-21-04-24--o3-price-reduction-high",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 40.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o3 (high)",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-25",
      "observedAtMin": "2025-06-25",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 60,
          "metricId": "pass_rate_1",
          "observedAt": "2025-06-27",
          "protocol": {
            "command": "aider --model o3",
            "edit_format": "architect",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 224
          },
          "rawValue": 34.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1570;dirname=2025-06-27-23-53-57--o3-mini-high-diff-arch",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 34.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o3 (high) + gpt-4.1",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-27",
      "observedAtMin": "2025-06-27",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 21,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-31",
          "protocol": {
            "command": "aider --model o3-mini --reasoning-effort high",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 224
          },
          "rawValue": 21.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=522;dirname=2025-01-31-20-42-47--o3-mini-diff-high",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 21.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o3-mini (high)",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-31",
      "observedAtMin": "2025-01-31",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 20,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-31",
          "protocol": {
            "command": "aider --model o3-mini",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 19.1,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=496;dirname=2025-01-31-20-27-46--o3-mini-diff2",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 19.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o3-mini (medium)",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-31",
      "observedAtMin": "2025-01-31",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 61,
          "metricId": "pass_rate_1",
          "observedAt": "2025-06-28",
          "protocol": {
            "command": "aider --model o3-pro",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 43.6,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1601;dirname=2025-06-28-00-38-18--o3-pro-high",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 43.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o3-pro (high)",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-28",
      "observedAtMin": "2025-06-28",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 43,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-16",
          "protocol": {
            "command": "aider --model o4-mini",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 19.6,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1096;dirname=2025-04-16-22-01-58--o4-mini-high-diff-exsys",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 19.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "o4-mini (high)",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-16",
      "observedAtMin": "2025-04-16",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 44,
          "metricId": "pass_rate_1",
          "observedAt": "2025-04-19",
          "protocol": {
            "command": "aider --model openrouter/all-hands/openhands-lm-32b-v0.1",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 4.0,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=1122;dirname=2025-04-19-14-43-04--o4-mini-patch",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 4.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "openhands-lm-32b-v0.1",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-19",
      "observedAtMin": "2025-04-19",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 19,
          "metricId": "pass_rate_1",
          "observedAt": "2025-01-28",
          "protocol": {
            "command": "OPENAI_API_BASE=https://dashscope-intl.aliyuncs.com/compatible-mode/v1 aider --model openai/qwen-max-2025-01-25",
            "edit_format": "diff",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 9.3,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=471;dirname=2025-01-28-16-00-03--qwen-max-2025-01-25-polyglot-diff",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 9.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "qwen-max-2025-01-25",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-28",
      "observedAtMin": "2025-01-28",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-aider-polyglot/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "line": 13,
          "metricId": "pass_rate_1",
          "observedAt": "2024-12-23",
          "protocol": {
            "command": "aider --model openai/yi-lightning",
            "edit_format": "whole",
            "harness": "aider",
            "subject_type": "system",
            "test_cases": 225
          },
          "rawValue": 5.8,
          "sourceId": "src-aider-polyglot",
          "sourceLabel": "Aider · Polyglot coding leaderboard",
          "sourceLocator": "yaml-row=313;dirname=2024-12-23-01-11-56--yi-test",
          "sourceUrl": "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml",
          "unit": "percent",
          "value": 5.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "pass_rate_1"
      ],
      "modelRef": "yi-lightning",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-23",
      "observedAtMin": "2024-12-23",
      "rowCount": 1,
      "sourceId": "src-aider-polyglot",
      "sourceIds": [
        "src-aider-polyglot"
      ],
      "sourceLabels": [
        "Aider · Polyglot coding leaderboard"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 80,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.10%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=79;rank=80;model=Amazon-Nova-2-Lite-v1:0 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Amazon-Nova-2-Lite-v1:0",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 95,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "22.29%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=94;rank=95;model=Amazon-Nova-Micro-v1:0 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 22.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Amazon-Nova-Micro-v1:0",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 88,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "24.97%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=87;rank=88;model=Amazon-Nova-Pro-v1:0 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 24.97
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Amazon-Nova-Pro-v1:0",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 60,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "32.14%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=59;rank=60;model=Arch-Agent-1.5B;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 32.14
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Arch-Agent-1.5B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 37,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "45.37%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=36;rank=37;model=Arch-Agent-32B;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 45.37
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Arch-Agent-32B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 56,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "35.36%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=55;rank=56;model=Arch-Agent-3B;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 35.36
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Arch-Agent-3B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 99,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "21.90%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=98;rank=99;model=Bielik-11B-v2.3-Instruct (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 21.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Bielik-11B-v2.3-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 36,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "46.23%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=35;rank=36;model=BitAgent-Bounty-8B;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 46.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "BitAgent-Bounty-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 74,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.99%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=73;rank=74;model=CoALM-70B;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.99
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "CoALM-70B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 84,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "unspecified",
            "calling_mode_label": null,
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "26.81%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=83;rank=84;model=CoALM-8B;mode=unknown",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 26.81
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "CoALM-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 35,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "46.49%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=34;rank=35;model=Command A (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 46.49
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Command A",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 13,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "57.06%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=12;rank=13;model=Command A Reasoning (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 57.06
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Command A Reasoning",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 61,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "32.07%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=60;rank=61;model=Command R7B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 32.07
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Command R7B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 82,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.01%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=81;rank=82;model=Falcon3-10B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.01
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Falcon3-10B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 106,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "11.08%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=105;rank=106;model=Falcon3-1B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 11.08
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Falcon3-1B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 104,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "16.25%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=103;rank=104;model=Falcon3-3B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 16.25
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Falcon3-3B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 91,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "24.03%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=90;rank=91;model=Falcon3-7B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 24.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Falcon3-7B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 4,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC thinking",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": "thinking"
          },
          "rawValue": "72.38%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=3;rank=4;model=GLM-4.6 (FC thinking);mode=FC thinking",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 72.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "GLM-4.6",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 66,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "30.43%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=65;rank=66;model=Gemma-3-12b-it (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 30.43
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Gemma-3-12b-it",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 109,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "7.17%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=108;rank=109;model=Gemma-3-1b-it (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 7.17
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Gemma-3-1b-it",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 69,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "29.47%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=68;rank=69;model=Gemma-3-27b-it (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 29.47
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Gemma-3-27b-it",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 101,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "19.62%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=100;rank=101;model=Gemma-3-4b-it (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 19.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Gemma-3-4b-it",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 93,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "23.23%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=92;rank=93;model=Granite-20b-FunctionCalling (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 23.23
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Granite-20b-FunctionCalling",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 81,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.10%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=80;rank=81;model=Granite-3.1-8B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Granite-3.1-8B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 83,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "26.87%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=82;rank=83;model=Granite-3.2-8B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 26.87
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Granite-3.2-8B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 103,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "18.98%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=102;rank=103;model=Granite-4.0-350m (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 18.98
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Granite-4.0-350m",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 12,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "58.29%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=11;rank=12;model=Grok-4-1-fast-non-reasoning (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 58.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Grok-4-1-fast-non-reasoning",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 5,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "69.57%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=4;rank=5;model=Grok-4-1-fast-reasoning (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 69.57
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Grok-4-1-fast-reasoning",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 100,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "21.22%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=99;rank=100;model=Hammer2.1-0.5b (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 21.22
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Hammer2.1-0.5b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 75,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.88%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=74;rank=75;model=Hammer2.1-1.5b (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.88
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Hammer2.1-1.5b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 68,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "29.71%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=67;rank=68;model=Hammer2.1-3b (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 29.71
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Hammer2.1-3b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 64,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "31.67%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=63;rank=64;model=Hammer2.1-7b (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 31.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Hammer2.1-7b",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 85,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "25.83%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=84;rank=85;model=Llama-3.1-8B-Instruct (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 25.83
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-3.1-8B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 108,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "10.00%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=107;rank=108;model=Llama-3.1-Nemotron-Ultra-253B-v1 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 10.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-3.1-Nemotron-Ultra-253B-v1",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 107,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "10.82%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=106;rank=107;model=Llama-3.2-1B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 10.82
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-3.2-1B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 98,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "21.95%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=97;rank=98;model=Llama-3.2-3B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 21.95
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-3.2-3B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 62,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "31.90%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=61;rank=62;model=Llama-3.3-70B-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 31.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-3.3-70B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 50,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "37.29%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=49;rank=50;model=Llama-4-Maverick-17B-128E-Instruct-FP8 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 37.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-4-Maverick-17B-128E-Instruct-FP8",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 72,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "28.13%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=71;rank=72;model=Llama-4-Scout-17B-16E-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 28.13
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Llama-4-Scout-17B-16E-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 97,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "22.08%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=96;rank=97;model=MiniCPM3-4B (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 22.08
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "MiniCPM3-4B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 86,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "25.55%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=85;rank=86;model=MiniCPM3-4B-FC (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 25.55
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "MiniCPM3-4B-FC",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 105,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "11.10%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=104;rank=105;model=Ministral-8B-Instruct-2410 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 11.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Ministral-8B-Instruct-2410",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 59,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "32.38%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=58;rank=59;model=Mistral-Small-2506 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 32.38
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Mistral-Small-2506",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 51,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "37.15%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=50;rank=51;model=Mistral-small-2506 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 37.15
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Mistral-small-2506",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 11,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "59.06%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=10;rank=11;model=Moonshotai-Kimi-K2-Instruct (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 59.06
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Moonshotai-Kimi-K2-Instruct",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 32,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "47.68%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=31;rank=32;model=Nanbeige3.5-Pro-Thinking (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 47.68
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Nanbeige3.5-Pro-Thinking",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 25,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "51.40%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=24;rank=25;model=Nanbeige4-3B-Thinking-2511 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 51.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Nanbeige4-3B-Thinking-2511",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 70,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "28.79%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=69;rank=70;model=Phi-4 (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 28.79
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Phi-4",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 71,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "28.41%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=70;rank=71;model=Qwen3-1.7B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 28.41
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "Qwen3-1.7B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 96,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "prompt",
            "calling_mode_label": "Prompt",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "22.25%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=95;rank=96;model=RZN-T (Prompt);mode=Prompt",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 22.25
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "RZN-T",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 40,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "42.44%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=39;rank=40;model=ToolACE-2-8B (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 42.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "ToolACE-2-8B",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 76,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "27.87%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=75;rank=76;model=palmyra-x-004 (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 27.87
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "palmyra-x-004",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 65,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "30.44%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=64;rank=65;model=xLAM-2-1b-fc-r (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 30.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "xLAM-2-1b-fc-r",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 18,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "54.66%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=17;rank=18;model=xLAM-2-32b-fc-r (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 54.66
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "xLAM-2-32b-fc-r",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 42,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "41.22%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=41;rank=42;model=xLAM-2-3b-fc-r (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 41.22
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "xLAM-2-3b-fc-r",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 22,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "53.07%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=21;rank=22;model=xLAM-2-70b-fc-r (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 53.07
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "xLAM-2-70b-fc-r",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "bfcl"
      ],
      "examples": [
        {
          "artifact": "src-bfcl/candidates.jsonl",
          "benchmarkId": "bfcl",
          "benchmarkName": "Berkeley Function Calling Leaderboard",
          "evidenceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
          "line": 34,
          "metricId": "accuracy",
          "observedAt": null,
          "protocol": {
            "benchmark_version": "BFCL-V4",
            "benchmark_version_id": "bfcl@v4",
            "calling_mode": "native_fc",
            "calling_mode_label": "FC",
            "commit": "f7cf735",
            "commit_url": "https://github.com/ShishirPatil/gorilla/commit/f7cf7359b7ac615a0b294831c5ba2bc95ee4a000",
            "evaluator": "bfcl",
            "evaluator_commit": "f7cf735",
            "harness": "bfcl-eval",
            "leaderboard_url": "https://gorilla.cs.berkeley.edu/leaderboard",
            "subject_type": "system",
            "variant": null
          },
          "rawValue": "46.68%",
          "sourceId": "src-bfcl",
          "sourceLabel": "Berkeley Function Calling Leaderboard",
          "sourceLocator": "csv-row=33;rank=34;model=xLAM-2-8b-fc-r (FC);mode=FC",
          "sourceUrl": "https://gorilla.cs.berkeley.edu/data_overall.csv",
          "unit": "percent",
          "value": 46.68
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:26:35Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "accuracy"
      ],
      "modelRef": "xLAM-2-8b-fc-r",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-bfcl",
      "sourceIds": [
        "src-bfcl"
      ],
      "sourceLabels": [
        "Berkeley Function Calling Leaderboard"
      ],
      "sourceUrls": [
        "https://gorilla.cs.berkeley.edu/data_overall.csv"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3064,
          "metricId": "EM",
          "observedAt": "2024-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.945",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=187;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 94.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "DeepSeek-Coder-V2-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-06-17",
      "observedAtMin": "2024-06-17",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3063,
          "metricId": "EM",
          "observedAt": "2024-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.876",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=186;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 87.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "DeepSeek-Coder-V2-Lite-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-06-13",
      "observedAtMin": "2024-06-13",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 69,
          "metricId": "Percent correct",
          "observedAt": "2024-12-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "17.8",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=59;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 17.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "DeepSeek-V2.5",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-21",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3440,
          "metricId": "Global average",
          "observedAt": "2024-09-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "52.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=31;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.64
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "Dracarys2-72B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-30",
      "observedAtMin": "2024-09-30",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3449,
          "metricId": "Global average",
          "observedAt": "2024-08-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "46.21",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=43;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.21
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "Dracarys2-Llama-3.1-70B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-08-14",
      "observedAtMin": "2024-08-14",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2936,
          "metricId": "mean_score",
          "observedAt": "2024-12-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3390151515151515",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=262;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 33.90151515151515
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "Eurus-2-7B-PRIME",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-31",
      "observedAtMin": "2024-12-31",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/81a60d18e010b27b36cd465c6604b915-Paper-Conference.pdf",
          "line": 4587,
          "metricId": "Score",
          "observedAt": "2024-08-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.701",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=100;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.701
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "InternVL-Chat-ViT-6B-Vicuna-13B",
      "numericRowCount": 1,
      "observedAtMax": "2024-08-16",
      "observedAtMin": "2024-08-16",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/81a60d18e010b27b36cd465c6604b915-Paper-Conference.pdf",
          "line": 4586,
          "metricId": "Score",
          "observedAt": "2023-12-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.662",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=99;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.662
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "InternVL-Chat-ViT-6B-Vicuna-7B",
      "numericRowCount": 1,
      "observedAtMax": "2023-12-25",
      "observedAtMin": "2023-12-25",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mindcube_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mindcube_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3629,
          "metricId": "Overall score",
          "observedAt": "2024-09-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.41960000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mindcube_external.csv:row=4;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 41.96
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Overall score"
      ],
      "modelRef": "LLaVA-Video-7B-Qwen2",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-02",
      "observedAtMin": "2024-09-02",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-balrog_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 895,
          "metricId": "Average progress",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.066",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=36;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 6.6000000000000005
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Average progress"
      ],
      "modelRef": "Llama-3.2-1B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-24",
      "observedAtMin": "2024-09-24",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-balrog_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 893,
          "metricId": "Average progress",
          "observedAt": "2024-09-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.10099999999999999",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=34;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 10.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Average progress"
      ],
      "modelRef": "Llama-3.2-3B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-24",
      "observedAtMin": "2024-09-24",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mindcube_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mindcube_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3630,
          "metricId": "Overall score",
          "observedAt": "2024-06-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.29460000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mindcube_external.csv:row=5;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.460000000000004
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Overall score"
      ],
      "modelRef": "LongVA-7B",
      "numericRowCount": 1,
      "observedAtMax": "2024-06-13",
      "observedAtMin": "2024-06-13",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4571,
          "metricId": "Score",
          "observedAt": "2024-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.694",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=84;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.694
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "MM1-3B-Chat",
      "numericRowCount": 1,
      "observedAtMax": "2024-03-14",
      "observedAtMin": "2024-03-14",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4572,
          "metricId": "Score",
          "observedAt": "2024-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.726",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=85;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.726
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "MM1-7B-Chat",
      "numericRowCount": 1,
      "observedAtMax": "2024-03-14",
      "observedAtMin": "2024-03-14",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3466,
          "metricId": "Global average",
          "observedAt": "2024-12-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "22.12",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=63;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 22.12
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "OLMo-2-1124-13B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-31",
      "observedAtMin": "2024-12-31",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4570,
          "metricId": "Score",
          "observedAt": "2024-08-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.913",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=83;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.913
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "Phi-3.5-vision-instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-08-16",
      "observedAtMin": "2024-08-16",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-geobench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-geobench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://geobench.org/",
          "line": 2675,
          "metricId": "ACW Avg Score",
          "observedAt": "2024-09-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "2131",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "geobench_external.csv:row=33;column=ACW Avg Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 2131.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ACW Avg Score"
      ],
      "modelRef": "Pixtral-12B-2409",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-17",
      "observedAtMin": "2024-09-17",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-lech_mazur_writing_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-lech_mazur_writing_external",
          "benchmarkName": null,
          "evidenceUrl": "https://github.com/lechmazur/Writing",
          "line": 3371,
          "metricId": "Mean score",
          "observedAt": "2025-03-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "8.02",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "lech_mazur_writing_external.csv:row=7;column=Mean score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 8.02
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Mean score"
      ],
      "modelRef": "QwQ-32B (16K thinking)",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-05",
      "observedAtMin": "2025-03-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-spatialviz_bench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-spatialviz_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4777,
          "metricId": "Overall score",
          "observedAt": "2024-01-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.32030000000000003",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "spatialviz_bench_external.csv:row=9;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 32.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Overall score"
      ],
      "modelRef": "Qwen-VL-Max",
      "numericRowCount": 1,
      "observedAtMax": "2024-01-18",
      "observedAtMin": "2024-01-18",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3822,
          "metricId": "EM",
          "observedAt": "2024-02-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.626",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=213;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 62.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "Qwen1.5-7B",
      "numericRowCount": 1,
      "observedAtMax": "2024-02-04",
      "observedAtMin": "2024-02-04",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-balrog_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 891,
          "metricId": "Average progress",
          "observedAt": "2024-08-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.128",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=32;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 12.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Average progress"
      ],
      "modelRef": "Qwen2-VL-72B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-08-29",
      "observedAtMin": "2024-08-29",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-balrog_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 896,
          "metricId": "Average progress",
          "observedAt": "2024-08-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.037000000000000005",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=37;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 3.7000000000000006
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Average progress"
      ],
      "modelRef": "Qwen2-VL-7B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-08-29",
      "observedAtMin": "2024-08-29",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3067,
          "metricId": "EM",
          "observedAt": "2024-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.942",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=190;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 94.19999999999999
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "Qwen2.5-Coder-14B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-11-06",
      "observedAtMin": "2024-11-06",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2409.12186",
          "line": 3065,
          "metricId": "EM",
          "observedAt": "2024-11-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.807",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=188;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 80.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "Qwen2.5-Coder-3B-Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2024-11-06",
      "observedAtMin": "2024-11-06",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2101.03961",
          "line": 4780,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.733",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=4;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.733
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "Switch-Base",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-superglue_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-superglue_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2101.03961",
          "line": 4782,
          "metricId": "Score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.847",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "superglue_external.csv:row=6;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.847
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "Switch-Large",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3849,
          "metricId": "EM",
          "observedAt": "2024-05-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.793",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=245;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "Yi-large",
      "numericRowCount": 1,
      "observedAtMax": "2024-05-13",
      "observedAtMin": "2024-05-13",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://scienceqa.github.io/leaderboard.html",
          "line": 4563,
          "metricId": "Score",
          "observedAt": "2023-02-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.7417",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=45;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.7417
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "blip2-opt-2.7b",
      "numericRowCount": 1,
      "observedAtMax": "2023-02-06",
      "observedAtMin": "2023-02-06",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 63,
          "metricId": "Percent correct",
          "observedAt": "2025-03-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "12.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=52;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 12.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "c4ai-command-a-03-2025",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-14",
      "observedAtMin": "2025-03-14",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 71,
          "metricId": "Percent correct",
          "observedAt": "2025-02-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "27.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=61;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 27.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "chatgpt-4o-01-29-2025",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-15",
      "observedAtMin": "2025-02-15",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 42,
          "metricId": "Percent correct",
          "observedAt": "2025-03-29",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "45.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=29;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 45.3
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "chatgpt-4o-03-27-2025",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-29",
      "observedAtMin": "2025-03-29",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2278,
          "metricId": "ECI Score",
          "observedAt": "2025-03-24",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "137.08",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=844;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 137.08
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/DeepSeek-V3-0324",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-24",
      "observedAtMin": "2025-03-24",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2279,
          "metricId": "ECI Score",
          "observedAt": "2025-03-11",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "130.83",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=845;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 130.83
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/Gemma-3-27b-It",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-11",
      "observedAtMin": "2025-03-11",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2282,
          "metricId": "ECI Score",
          "observedAt": "2025-04-06",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "130.52",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=855;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 130.52
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/Llama-4-Scout-17B-16E Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-06",
      "observedAtMin": "2025-04-06",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2275,
          "metricId": "ECI Score",
          "observedAt": "2025-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "139.64",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=839;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 139.64
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/Qwen3-235B-A22B",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-28",
      "observedAtMin": "2025-04-28",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2276,
          "metricId": "ECI Score",
          "observedAt": "2025-07-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.77",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=840;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.77
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/Qwen3-235B-A22B-Thinking-2507",
      "numericRowCount": 1,
      "observedAtMax": "2025-07-25",
      "observedAtMin": "2025-07-25",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2272,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=835;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/gpt-oss-120b",
      "numericRowCount": 1,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2273,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "140.69",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=836;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 140.69
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "chutes/gpt-oss-120b_high",
      "numericRowCount": 1,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-metr_time_horizons_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3577,
          "metricId": "average_score",
          "observedAt": "2026-04-07",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.85205",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=2;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.85205
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "average_score"
      ],
      "modelRef": "claude-mythos-preview-early",
      "numericRowCount": 1,
      "observedAtMax": "2026-04-07",
      "observedAtMin": "2026-04-07",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1882,
          "metricId": "ECI Score",
          "observedAt": "2025-08-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "144.44",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=175;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 144.44
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "claude-opus-4-1-20250805_32K",
      "numericRowCount": 1,
      "observedAtMax": "2025-08-05",
      "observedAtMin": "2025-08-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-gsm8k_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gsm8k_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/classic/latest/#/groups/gsm",
          "line": 2941,
          "metricId": "EM",
          "observedAt": "2022-03-15",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.568",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gsm8k_external.csv:row=3;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 56.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "code-davinci-002",
      "numericRowCount": 1,
      "observedAtMax": "2022-03-15",
      "observedAtMin": "2022-03-15",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 72,
          "metricId": "Percent correct",
          "observedAt": "2025-01-13",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "11.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=62;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 11.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "codestral-2501",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-13",
      "observedAtMin": "2025-01-13",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-ale_bench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 197,
          "metricId": "Performance",
          "observedAt": "2025-07-30",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "137.78",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=111;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 137.78
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Performance"
      ],
      "modelRef": "codestral-2508",
      "numericRowCount": 1,
      "observedAtMax": "2025-07-30",
      "observedAtMin": "2025-07-30",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-metr_time_horizons_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "line": 3625,
          "metricId": "average_score",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.161869",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=50;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.161869
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "average_score"
      ],
      "modelRef": "davinci-002",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 4579,
          "metricId": "Score",
          "observedAt": "2024-05-21",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.749",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=92;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.749
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "falcon-11B-vlm",
      "numericRowCount": 1,
      "observedAtMax": "2024-05-21",
      "observedAtMin": "2024-05-21",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-frontiermath_tier_4"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2503,
          "metricId": "mean_score",
          "observedAt": "2026-05-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.479",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=10;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 47.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "gdm-ai-co-mathematician",
      "numericRowCount": 1,
      "observedAtMax": "2026-05-08",
      "observedAtMin": "2026-05-08",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4577,
          "metricId": "Score",
          "observedAt": "2024-01-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.797",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=90;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.797
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "gemini-1.0-pro-vision",
      "numericRowCount": 1,
      "observedAtMax": "2024-01-04",
      "observedAtMin": "2024-01-04",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3438,
          "metricId": "Global average",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "54.29",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=29;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 54.29
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "gemini-2.0-flash-lite",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-05",
      "observedAtMin": "2025-02-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-forecastbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2352,
          "metricId": "Overall score",
          "observedAt": "2025-02-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.1",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=69;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.1
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Overall score"
      ],
      "modelRef": "gemini-2.0-flash-lite-001",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-25",
      "observedAtMin": "2025-02-25",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3439,
          "metricId": "Global average",
          "observedAt": "2025-02-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "53.24",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=30;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 53.24
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "gemini-2.0-flash-lite-preview-02-05",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-05",
      "observedAtMin": "2025-02-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-weirdml_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-weirdml_external",
          "benchmarkName": null,
          "evidenceUrl": "https://htihle.github.io/weirdml.html",
          "line": 5527,
          "metricId": "Accuracy",
          "observedAt": "2025-06-17",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3522",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "weirdml_external.csv:row=134;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 35.22
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Accuracy"
      ],
      "modelRef": "gemini-2.5-flash-lite-preview-06-17_16K",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-17",
      "observedAtMin": "2025-06-17",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-vpct_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-vpct_external",
          "benchmarkName": null,
          "evidenceUrl": "https://cbrower.dev/vpct",
          "line": 5277,
          "metricId": "Correct",
          "observedAt": "2025-09-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.3",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "vpct_external.csv:row=39;column=Correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 30.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Correct"
      ],
      "modelRef": "gemini-2.5-flash-lite-preview-09-2025",
      "numericRowCount": 1,
      "observedAtMax": "2025-09-25",
      "observedAtMin": "2025-09-25",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-vpct_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-vpct_external",
          "benchmarkName": null,
          "evidenceUrl": "https://cbrower.dev/vpct",
          "line": 5254,
          "metricId": "Correct",
          "observedAt": "2025-09-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.408",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "vpct_external.csv:row=16;column=Correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 40.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Correct"
      ],
      "modelRef": "gemini-robotics-er-1.5-preview",
      "numericRowCount": 1,
      "observedAtMax": "2025-09-26",
      "observedAtMin": "2025-09-26",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-blueprint_bench_2_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-blueprint_bench_2_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1005,
          "metricId": "Score",
          "observedAt": "2026-04-14",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "blueprint_bench_2_external.csv:row=22;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "gemini-robotics-er-1.6-preview",
      "numericRowCount": 1,
      "observedAtMax": "2026-04-14",
      "observedAtMin": "2026-04-14",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-metr_time_horizons_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-metr_time_horizons_external",
          "benchmarkName": null,
          "evidenceUrl": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "line": 3624,
          "metricId": "average_score",
          "observedAt": "2023-09-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.214611",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "metr_time_horizons_external.csv:row=49;column=average_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 0.214611
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "average_score"
      ],
      "modelRef": "gpt-3.5-turbo-instruct",
      "numericRowCount": 1,
      "observedAtMax": "2023-09-18",
      "observedAtMin": "2023-09-18",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2280,
          "metricId": "ECI Score",
          "observedAt": "2025-04-09",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "141.04",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=847;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 141.04
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "grok-3-mini-beta_medium",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-09",
      "observedAtMin": "2025-04-09",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-simplebench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-simplebench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 4628,
          "metricId": "Score (AVG@5)",
          "observedAt": "2025-11-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.56",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "simplebench_external.csv:row=41;column=Score (AVG@5)",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.56
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score (AVG@5)"
      ],
      "modelRef": "grok-4-1-fast-non-reasoning",
      "numericRowCount": 1,
      "observedAtMax": "2025-11-19",
      "observedAtMin": "2025-11-19",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-frontiermath_tier_4"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiermath_tier_4",
          "benchmarkName": "FrontierMath Tier 4 v2 (reported snapshot)",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2545,
          "metricId": "mean_score",
          "observedAt": "2025-07-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0208",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath_tier_4.csv:row=52;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 2.08
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "grok-4-heavy-web-app",
      "numericRowCount": 1,
      "observedAtMax": "2025-07-10",
      "observedAtMin": "2025-07-10",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/81a60d18e010b27b36cd465c6604b915-Paper-Conference.pdf",
          "line": 4584,
          "metricId": "Score",
          "observedAt": "2023-12-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.631",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=97;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.631
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "instructblip-vicuna-13b",
      "numericRowCount": 1,
      "observedAtMax": "2023-12-25",
      "observedAtMin": "2023-12-25",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/81a60d18e010b27b36cd465c6604b915-Paper-Conference.pdf",
          "line": 4583,
          "metricId": "Score",
          "observedAt": "2023-05-22",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.605",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=96;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.605
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "instructblip-vicuna-7b",
      "numericRowCount": 1,
      "observedAtMax": "2023-05-22",
      "observedAtMin": "2023-05-22",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3442,
          "metricId": "Global average",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "52.19",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=34;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 52.19
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "learnlm-1.5-pro-experimental",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2404.14219",
          "line": 4574,
          "metricId": "Score",
          "observedAt": "2024-04-20",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.737",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=87;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.737
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "llama3-llava-next-8b",
      "numericRowCount": 1,
      "observedAtMax": "2024-04-20",
      "observedAtMin": "2024-04-20",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/81a60d18e010b27b36cd465c6604b915-Paper-Conference.pdf",
          "line": 4588,
          "metricId": "Score",
          "observedAt": "2023-10-05",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.668",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=101;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.668
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "llava-v1.5-7b",
      "numericRowCount": 1,
      "observedAtMax": "2023-10-05",
      "observedAtMin": "2023-10-05",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 4582,
          "metricId": "Score",
          "observedAt": "2024-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.728",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=95;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.728
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "llava-v1.6-mistral-7b",
      "numericRowCount": 1,
      "observedAtMax": "2024-01-31",
      "observedAtMin": "2024-01-31",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-science_qa_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-science_qa_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2407.14885",
          "line": 4581,
          "metricId": "Score",
          "observedAt": "2024-01-31",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.736",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "science_qa_external.csv:row=94;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.736
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "llava-v1.6-vicuna-13b",
      "numericRowCount": 1,
      "observedAtMax": "2024-01-31",
      "observedAtMin": "2024-01-31",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mindcube_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mindcube_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 3628,
          "metricId": "Overall score",
          "observedAt": "2024-11-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.4485",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mindcube_external.csv:row=3;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 44.85
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Overall score"
      ],
      "modelRef": "mPLUG-Owl3-7B-241101",
      "numericRowCount": 1,
      "observedAtMax": "2024-11-26",
      "observedAtMin": "2024-11-26",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3787,
          "metricId": "EM",
          "observedAt": "2024-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.687",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=168;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "mistral-small-2402",
      "numericRowCount": 1,
      "observedAtMax": "2024-02-26",
      "observedAtMin": "2024-02-26",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-ale_bench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 177,
          "metricId": "Performance",
          "observedAt": "2026-03-16",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "497.62",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=91;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 497.62
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Performance"
      ],
      "modelRef": "mistral-small-2603",
      "numericRowCount": 1,
      "observedAtMax": "2026-03-16",
      "observedAtMin": "2026-03-16",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-gdp_pdf_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-gdp_pdf_external",
          "benchmarkName": null,
          "evidenceUrl": "https://surgehq.ai/benchmarks/gdp-pdf",
          "line": 2630,
          "metricId": "GDP.pdf score",
          "observedAt": "2025-12-02",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.02",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gdp_pdf_external.csv:row=28;column=GDP.pdf score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.02
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "GDP.pdf score"
      ],
      "modelRef": "nova-2.0-pro-preview_unknown",
      "numericRowCount": 1,
      "observedAtMax": "2025-12-02",
      "observedAtMin": "2025-12-02",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-arc_agi_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-arc_agi_external",
          "benchmarkName": null,
          "evidenceUrl": "https://arcprize.org/leaderboard",
          "line": 679,
          "metricId": "Score",
          "observedAt": "2025-03-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.233",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "arc_agi_external.csv:row=190;column=Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.233
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Score"
      ],
      "modelRef": "o1-pro-2025-03-19_low",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-19",
      "observedAtMin": "2025-03-19",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 25,
          "metricId": "Percent correct",
          "observedAt": "2025-04-19",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "10.2",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=12;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 10.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "openhands-lm-32b-v0.1",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-19",
      "observedAtMin": "2025-04-19",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-epoch_capabilities_index"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-epoch_capabilities_index",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2281,
          "metricId": "ECI Score",
          "observedAt": "2025-09-01",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "139.03",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "epoch_capabilities_index.csv:row=850;column=ECI Score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "index",
          "value": 139.03
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "ECI Score"
      ],
      "modelRef": "parasail-qwen3-235b-a22b-instruct-2507",
      "numericRowCount": 1,
      "observedAtMax": "2025-09-01",
      "observedAtMin": "2025-09-01",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-enigma_eval_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-enigma_eval_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 1706,
          "metricId": "Accuracy",
          "observedAt": "2024-11-18",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.0084",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "enigma_eval_external.csv:row=37;column=Accuracy",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 0.84
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Accuracy"
      ],
      "modelRef": "pixtral-large-2411",
      "numericRowCount": 1,
      "observedAtMax": "2024-11-18",
      "observedAtMin": "2024-11-18",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "frontiermath"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "frontiermath",
          "benchmarkName": "FrontierMath T1–T3",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2465,
          "metricId": "mean_score",
          "observedAt": "2025-04-28",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.017241379310344827",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiermath.csv:row=73;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 1.7241379310344827
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwen-plus-2025-04-28",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-28",
      "observedAtMin": "2025-04-28",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-forecastbench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-forecastbench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2345,
          "metricId": "Overall score",
          "observedAt": "2024-04-25",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "57.7",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "forecastbench_external.csv:row=62;column=Overall score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 57.7
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Overall score"
      ],
      "modelRef": "qwen1.5-110b-chat",
      "numericRowCount": 1,
      "observedAtMax": "2024-04-25",
      "observedAtMin": "2024-04-25",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3823,
          "metricId": "EM",
          "observedAt": "2024-02-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.686",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=214;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 68.60000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "qwen1.5-14B",
      "numericRowCount": 1,
      "observedAtMax": "2024-02-04",
      "observedAtMin": "2024-02-04",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "https://crfm.stanford.edu/helm/lite/latest/#/leaderboard/mmlu",
          "line": 3824,
          "metricId": "EM",
          "observedAt": "2024-02-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.744",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=215;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 74.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "qwen1.5-32B",
      "numericRowCount": 1,
      "observedAtMax": "2024-02-04",
      "observedAtMin": "2024-02-04",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2914,
          "metricId": "mean_score",
          "observedAt": "2024-04-03",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.307449494949495",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=240;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 30.7449494949495
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwen1.5-32b-chat",
      "numericRowCount": 1,
      "observedAtMax": "2024-04-03",
      "observedAtMin": "2024-04-03",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2913,
          "metricId": "mean_score",
          "observedAt": "2024-02-04",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.2881944444444444",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=239;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 28.819444444444443
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwen1.5-72b-chat",
      "numericRowCount": 1,
      "observedAtMax": "2024-02-04",
      "observedAtMin": "2024-02-04",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-mmlu_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-mmlu_external",
          "benchmarkName": null,
          "evidenceUrl": "http://arxiv.org/abs/2412.08905",
          "line": 3826,
          "metricId": "EM",
          "observedAt": "2025-02-26",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.799",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "mmlu_external.csv:row=219;column=EM",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 79.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "EM"
      ],
      "modelRef": "qwen2.5-14b-instruct",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-26",
      "observedAtMin": "2025-02-26",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-ale_bench_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-ale_bench_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 179,
          "metricId": "Performance",
          "observedAt": "2025-09-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "456.5",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "ale_bench_external.csv:row=93;column=Performance",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "score",
          "value": 456.5
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Performance"
      ],
      "modelRef": "qwen3-coder-plus",
      "numericRowCount": 1,
      "observedAtMax": "2025-09-23",
      "observedAtMin": "2025-09-23",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "gpqa-diamond"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "gpqa-diamond",
          "benchmarkName": "GPQA Diamond",
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2851,
          "metricId": "mean_score",
          "observedAt": "2025-04-08",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.6540404040404041",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "gpqa_diamond.csv:row=177;column=mean_score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 65.40404040404042
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "mean_score"
      ],
      "modelRef": "qwq-plus",
      "numericRowCount": 1,
      "observedAtMax": "2025-04-08",
      "observedAtMin": "2025-04-08",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-balrog_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-balrog_external",
          "benchmarkName": null,
          "evidenceUrl": "https://balrogai.com/",
          "line": 878,
          "metricId": "Average progress",
          "observedAt": "2025-03-10",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "0.292",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "balrog_external.csv:row=19;column=Average progress",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 29.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Average progress"
      ],
      "modelRef": "reka-flash-3",
      "numericRowCount": 1,
      "observedAtMax": "2025-03-10",
      "observedAtMin": "2025-03-10",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "livebench"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "livebench",
          "benchmarkName": "LiveBench",
          "evidenceUrl": "https://livebench.ai/#/",
          "line": 3447,
          "metricId": "Global average",
          "observedAt": null,
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "46.88",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "live_bench_external.csv:row=41;column=Global average",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 46.88
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Global average"
      ],
      "modelRef": "sonar",
      "numericRowCount": 1,
      "observedAtMax": null,
      "observedAtMin": null,
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-frontiercode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiercode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2392,
          "metricId": "Main score",
          "observedAt": "2026-04-07",
          "protocol": {
            "harness": "chisel",
            "subject_type": "system"
          },
          "rawValue": "0.09390000000000001",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiercode_external.csv:row=33;column=Main score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.09390000000000001
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Main score"
      ],
      "modelRef": "swe-1.6",
      "numericRowCount": 1,
      "observedAtMax": "2026-04-07",
      "observedAtMin": "2026-04-07",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "epoch-frontiercode_external"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "epoch-frontiercode_external",
          "benchmarkName": null,
          "evidenceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "line": 2376,
          "metricId": "Main score",
          "observedAt": "2026-07-08",
          "protocol": {
            "harness": "chisel",
            "subject_type": "system"
          },
          "rawValue": "0.41990000000000005",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "frontiercode_external.csv:row=13;column=Main score",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "fraction",
          "value": 0.41990000000000005
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Main score"
      ],
      "modelRef": "swe-1.7",
      "numericRowCount": 1,
      "observedAtMax": "2026-07-08",
      "observedAtMin": "2026-07-08",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "aider-polyglot"
      ],
      "examples": [
        {
          "artifact": "src-epoch-benchmark-hub/candidates.jsonl",
          "benchmarkId": "aider-polyglot",
          "benchmarkName": "Aider Polyglot",
          "evidenceUrl": "https://aider.chat/docs/leaderboards/#polyglot-leaderboard",
          "line": 75,
          "metricId": "Percent correct",
          "observedAt": "2024-12-23",
          "protocol": {
            "harness": null,
            "subject_type": "model"
          },
          "rawValue": "12.9",
          "sourceId": "src-epoch-benchmark-hub",
          "sourceLabel": "Epoch AI · Benchmarking Hub downloadable snapshot",
          "sourceLocator": "aider_polyglot_external.csv:row=65;column=Percent correct",
          "sourceUrl": "https://epoch.ai/data/benchmark_data.zip",
          "unit": "percent",
          "value": 12.9
        }
      ],
      "latestRetrievedAt": "2026-08-27T08:25:00Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "Percent correct"
      ],
      "modelRef": "yi-lightning",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-23",
      "observedAtMin": "2024-12-23",
      "rowCount": 1,
      "sourceId": "src-epoch-benchmark-hub",
      "sourceIds": [
        "src-epoch-benchmark-hub"
      ],
      "sourceLabels": [
        "Epoch AI · Benchmarking Hub downloadable snapshot"
      ],
      "sourceUrls": [
        "https://epoch.ai/data/benchmark_data.zip"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 159,
          "metricId": "resolved",
          "observedAt": "2025-02-03",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Verified",
            "scaffold": "OpenHands",
            "subject_type": "system"
          },
          "rawValue": 60.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[74]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 60.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "4x Scaled",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-03",
      "observedAtMin": "2025-02-03",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 211,
          "metricId": "resolved",
          "observedAt": "2025-05-27",
          "protocol": {
            "checked": false,
            "harness": "Amazon Nova Premier 1.0",
            "leaderboard_variant": "Verified",
            "scaffold": "Amazon Nova Premier 1.0",
            "subject_type": "system"
          },
          "rawValue": 42.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[126]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 42.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Amazon.nova Premier v1:0",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-27",
      "observedAtMin": "2025-05-27",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 219,
          "metricId": "resolved",
          "observedAt": "2024-10-22",
          "protocol": {
            "checked": false,
            "harness": "Tools",
            "leaderboard_variant": "Verified",
            "scaffold": "Tools",
            "subject_type": "system"
          },
          "rawValue": 40.6,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[134]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 40.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 3.5 Haiku",
      "numericRowCount": 1,
      "observedAtMax": "2024-10-22",
      "observedAtMin": "2024-10-22",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 156,
          "metricId": "resolved",
          "observedAt": "2025-02-25",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "SWE-agent",
            "subject_type": "system"
          },
          "rawValue": 62.4,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[71]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 62.4
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Claude 3.7 Sonnet w/ Review Heavy",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-25",
      "observedAtMin": "2025-02-25",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 203,
          "metricId": "resolved",
          "observedAt": "2025-05-28",
          "protocol": {
            "checked": false,
            "harness": "PatchPilot",
            "leaderboard_variant": "Verified",
            "scaffold": "PatchPilot",
            "subject_type": "model"
          },
          "rawValue": 46.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[118]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 46.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Co-PatcheR",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-28",
      "observedAtMin": "2025-05-28",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 325,
          "metricId": "resolved",
          "observedAt": "2024-07-25",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Lite",
            "scaffold": "OpenHands",
            "subject_type": "system"
          },
          "rawValue": 26.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[60]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 26.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "CodeAct v1.8",
      "numericRowCount": 1,
      "observedAtMax": "2024-07-25",
      "observedAtMin": "2024-07-25",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 213,
          "metricId": "resolved",
          "observedAt": "2025-06-29",
          "protocol": {
            "checked": false,
            "harness": "R2E-Gym",
            "leaderboard_variant": "Verified",
            "scaffold": "R2E-Gym",
            "subject_type": "model"
          },
          "rawValue": 42.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[128]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 42.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "DeepSWE-Preview",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-29",
      "observedAtMin": "2025-06-29",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 300,
          "metricId": "resolved",
          "observedAt": "2025-06-09",
          "protocol": {
            "checked": false,
            "harness": "KGCompass",
            "leaderboard_variant": "Lite",
            "scaffold": "KGCompass",
            "subject_type": "model"
          },
          "rawValue": 36.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[35]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 36.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "DeepSeek V3",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-09",
      "observedAtMin": "2025-06-09",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 214,
          "metricId": "resolved",
          "observedAt": "2025-08-06",
          "protocol": {
            "checked": null,
            "harness": "SWE-Exp",
            "leaderboard_variant": "Verified",
            "scaffold": "SWE-Exp",
            "subject_type": "model"
          },
          "rawValue": 42.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[129]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 42.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "DeepSeek V3 0324",
      "numericRowCount": 1,
      "observedAtMax": "2025-08-06",
      "observedAtMin": "2025-08-06",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 312,
          "metricId": "resolved",
          "observedAt": "2025-01-11",
          "protocol": {
            "checked": true,
            "harness": "Moatless Tools",
            "leaderboard_variant": "Lite",
            "scaffold": "Moatless Tools",
            "subject_type": "model"
          },
          "rawValue": 30.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[47]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 30.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Deepseek V3",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-11",
      "observedAtMin": "2025-01-11",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 200,
          "metricId": "resolved",
          "observedAt": "2025-05-20",
          "protocol": {
            "checked": true,
            "harness": "OpenHands",
            "leaderboard_variant": "Verified",
            "scaffold": "OpenHands",
            "subject_type": "model"
          },
          "rawValue": 46.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[115]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 46.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "DevStral Small 2505",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-20",
      "observedAtMin": "2025-05-20",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 229,
          "metricId": "resolved",
          "observedAt": "2025-07-25",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 38.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[144]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 38.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "DevStral Small 2507",
      "numericRowCount": 1,
      "observedAtMax": "2025-07-25",
      "observedAtMin": "2025-07-25",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 87,
          "metricId": "resolved",
          "observedAt": "2025-09-28",
          "protocol": {
            "checked": false,
            "harness": "TRAE",
            "leaderboard_variant": "Verified",
            "scaffold": "TRAE",
            "subject_type": "system"
          },
          "rawValue": 78.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[2]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 78.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Doubao-Seed-Code",
      "numericRowCount": 1,
      "observedAtMax": "2025-09-28",
      "observedAtMin": "2025-09-28",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 181,
          "metricId": "resolved",
          "observedAt": "2025-11-10",
          "protocol": {
            "checked": null,
            "harness": "FrogBoss-32B-2510",
            "leaderboard_variant": "Verified",
            "scaffold": "FrogBoss-32B-2510",
            "subject_type": "system"
          },
          "rawValue": 53.6,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[96]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 53.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Frogboss 32B 2510",
      "numericRowCount": 1,
      "observedAtMax": "2025-11-10",
      "observedAtMin": "2025-11-10",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 206,
          "metricId": "resolved",
          "observedAt": "2025-11-10",
          "protocol": {
            "checked": null,
            "harness": "FrogMini-14B-2510",
            "leaderboard_variant": "Verified",
            "scaffold": "FrogMini-14B-2510",
            "subject_type": "system"
          },
          "rawValue": 45.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[121]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 45.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Frogmini 14B 2510",
      "numericRowCount": 1,
      "observedAtMax": "2025-11-10",
      "observedAtMin": "2025-11-10",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 150,
          "metricId": "resolved",
          "observedAt": "2025-07-28",
          "protocol": {
            "checked": false,
            "harness": "Undisclosed",
            "leaderboard_variant": "Verified",
            "scaffold": "Undisclosed",
            "subject_type": "model"
          },
          "rawValue": 64.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[65]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 64.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GLM-4.5",
      "numericRowCount": 1,
      "observedAtMax": "2025-07-28",
      "observedAtMin": "2025-07-28",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 135,
          "metricId": "resolved",
          "observedAt": "2025-09-30",
          "protocol": {
            "checked": false,
            "harness": "Undisclosed",
            "leaderboard_variant": "Verified",
            "scaffold": "Undisclosed",
            "subject_type": "model"
          },
          "rawValue": 68.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[50]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 68.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GLM-4.6",
      "numericRowCount": 1,
      "observedAtMax": "2025-09-30",
      "observedAtMin": "2025-09-30",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 338,
          "metricId": "resolved",
          "observedAt": "2024-05-30",
          "protocol": {
            "checked": false,
            "harness": "AutoCodeRover",
            "leaderboard_variant": "Lite",
            "scaffold": "AutoCodeRover",
            "subject_type": "system"
          },
          "rawValue": 19.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[73]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 19.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4 (0125)",
      "numericRowCount": 1,
      "observedAtMax": "2024-05-30",
      "observedAtMin": "2024-05-30",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 332,
          "metricId": "resolved",
          "observedAt": "2024-05-24",
          "protocol": {
            "checked": false,
            "harness": "OpenCSG StarShip CodeGenAgent",
            "leaderboard_variant": "Lite",
            "scaffold": "OpenCSG StarShip CodeGenAgent",
            "subject_type": "system"
          },
          "rawValue": 23.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[67]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 23.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4 (0613)",
      "numericRowCount": 1,
      "observedAtMax": "2024-05-24",
      "observedAtMin": "2024-05-24",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 293,
          "metricId": "resolved",
          "observedAt": "2025-01-13",
          "protocol": {
            "checked": false,
            "harness": "OpenCSG Starship Agentic Coder",
            "leaderboard_variant": "Lite",
            "scaffold": "OpenCSG Starship Agentic Coder",
            "subject_type": "system"
          },
          "rawValue": 39.67,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[28]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 39.67
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4 (0806)",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-13",
      "observedAtMin": "2025-01-13",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-multimodal"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-multimodal",
          "benchmarkName": "SWE-bench Multimodal",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 355,
          "metricId": "resolved",
          "observedAt": "2025-05-31",
          "protocol": {
            "checked": true,
            "harness": "GUIRepair",
            "leaderboard_variant": "Multimodal",
            "scaffold": "GUIRepair",
            "subject_type": "system"
          },
          "rawValue": 31.14,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[5]=Multimodal;results[6]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 31.14
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4.1",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-31",
      "observedAtMin": "2025-05-31",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 327,
          "metricId": "resolved",
          "observedAt": "2024-05-23",
          "protocol": {
            "checked": false,
            "harness": "Aider",
            "leaderboard_variant": "Lite",
            "scaffold": "Aider",
            "subject_type": "system"
          },
          "rawValue": 26.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[62]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 26.33
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "GPT-4o & Claude 3 Opus",
      "numericRowCount": 1,
      "observedAtMax": "2024-05-23",
      "observedAtMin": "2024-05-23",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 208,
          "metricId": "resolved",
          "observedAt": "2025-01-18",
          "protocol": {
            "checked": false,
            "harness": "CodeShellAgent",
            "leaderboard_variant": "Verified",
            "scaffold": "CodeShellAgent",
            "subject_type": "system"
          },
          "rawValue": 44.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[123]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 44.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Gemini 2.0 Flash (Experimental)",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-18",
      "observedAtMin": "2025-01-18",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 188,
          "metricId": "resolved",
          "observedAt": "2024-12-12",
          "protocol": {
            "checked": false,
            "harness": "Google Jules",
            "leaderboard_variant": "Verified",
            "scaffold": "Google Jules",
            "subject_type": "system"
          },
          "rawValue": 52.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[103]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 52.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Gemini 2.0 Flash (v20241212-experimental)",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-12",
      "observedAtMin": "2024-12-12",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 245,
          "metricId": "resolved",
          "observedAt": "2024-09-18",
          "protocol": {
            "checked": false,
            "harness": "Lingma Agent",
            "leaderboard_variant": "Verified",
            "scaffold": "Lingma Agent",
            "subject_type": "system"
          },
          "rawValue": 25.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[160]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 25.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Lingma SWE-GPT 72b (v0918)",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 238,
          "metricId": "resolved",
          "observedAt": "2024-10-02",
          "protocol": {
            "checked": false,
            "harness": "Lingma Agent",
            "leaderboard_variant": "Verified",
            "scaffold": "Lingma Agent",
            "subject_type": "system"
          },
          "rawValue": 28.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[153]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 28.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Lingma SWE-GPT 72b (v0925)",
      "numericRowCount": 1,
      "observedAtMax": "2024-10-02",
      "observedAtMin": "2024-10-02",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 256,
          "metricId": "resolved",
          "observedAt": "2024-09-18",
          "protocol": {
            "checked": false,
            "harness": "Lingma Agent",
            "leaderboard_variant": "Verified",
            "scaffold": "Lingma Agent",
            "subject_type": "system"
          },
          "rawValue": 10.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[171]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 10.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Lingma SWE-GPT 7b (v0918)",
      "numericRowCount": 1,
      "observedAtMax": "2024-09-18",
      "observedAtMin": "2024-09-18",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 253,
          "metricId": "resolved",
          "observedAt": "2024-10-02",
          "protocol": {
            "checked": false,
            "harness": "Lingma Agent",
            "leaderboard_variant": "Verified",
            "scaffold": "Lingma Agent",
            "subject_type": "system"
          },
          "rawValue": 18.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[168]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 18.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Lingma SWE-GPT 7b (v0925)",
      "numericRowCount": 1,
      "observedAtMax": "2024-10-02",
      "observedAtMin": "2024-10-02",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 217,
          "metricId": "resolved",
          "observedAt": "2025-02-26",
          "protocol": {
            "checked": false,
            "harness": "Agentless Mini",
            "leaderboard_variant": "Verified",
            "scaffold": "Agentless Mini",
            "subject_type": "system"
          },
          "rawValue": 41.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[132]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 41.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Llama3-SWE-RL-70B",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-26",
      "observedAtMin": "2025-02-26",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 285,
          "metricId": "resolved",
          "observedAt": "2024-07-02",
          "protocol": {
            "checked": false,
            "harness": "CodeStory Aide",
            "leaderboard_variant": "Lite",
            "scaffold": "CodeStory Aide",
            "subject_type": "system"
          },
          "rawValue": 43.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[20]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 43.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Mixed Models",
      "numericRowCount": 1,
      "observedAtMax": "2024-07-02",
      "observedAtMin": "2024-07-02",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 228,
          "metricId": "resolved",
          "observedAt": "2025-06-16",
          "protocol": {
            "checked": true,
            "harness": "Skywork-SWE-32B",
            "leaderboard_variant": "Verified",
            "scaffold": "Skywork-SWE-32B",
            "subject_type": "model"
          },
          "rawValue": 38.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[143]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 38.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "Qwen2.5 Coder 32B Instruct",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-16",
      "observedAtMin": "2025-06-16",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 222,
          "metricId": "resolved",
          "observedAt": "2025-05-11",
          "protocol": {
            "checked": true,
            "harness": "SWE-agent",
            "leaderboard_variant": "Verified",
            "scaffold": "SWE-agent",
            "subject_type": "model"
          },
          "rawValue": 40.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[137]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 40.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "SWE-agent-LM-32B",
      "numericRowCount": 1,
      "observedAtMax": "2025-05-11",
      "observedAtMin": "2025-05-11",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 164,
          "metricId": "resolved",
          "observedAt": "2025-06-29",
          "protocol": {
            "checked": false,
            "harness": "DeepSWE-Preview",
            "leaderboard_variant": "Verified",
            "scaffold": "DeepSWE-Preview",
            "subject_type": "model"
          },
          "rawValue": 58.8,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[79]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 58.8
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "TTS(Bo16)",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-29",
      "observedAtMin": "2025-06-29",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 199,
          "metricId": "resolved",
          "observedAt": "2025-06-16",
          "protocol": {
            "checked": true,
            "harness": "Skywork-SWE-32B",
            "leaderboard_variant": "Verified",
            "scaffold": "Skywork-SWE-32B",
            "subject_type": "model"
          },
          "rawValue": 47.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[114]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 47.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "TTS(Bo8)",
      "numericRowCount": 1,
      "observedAtMax": "2025-06-16",
      "observedAtMin": "2025-06-16",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 316,
          "metricId": "resolved",
          "observedAt": "2024-12-03",
          "protocol": {
            "checked": true,
            "harness": "Kortix AI",
            "leaderboard_variant": "Lite",
            "scaffold": "Kortix AI",
            "subject_type": "system"
          },
          "rawValue": 30.0,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[51]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 30.0
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "claude-3-5-sonnet-20241022",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-03",
      "observedAtMin": "2024-12-03",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 148,
          "metricId": "resolved",
          "observedAt": "2025-01-17",
          "protocol": {
            "checked": false,
            "harness": "W&B Programmer O1 crosscheck5",
            "leaderboard_variant": "Verified",
            "scaffold": "W&B Programmer O1 crosscheck5",
            "subject_type": "system"
          },
          "rawValue": 64.6,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[63]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 64.6
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "o1-preview",
      "numericRowCount": 1,
      "observedAtMax": "2025-01-17",
      "observedAtMin": "2025-01-17",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-lite"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-lite",
          "benchmarkName": "SWE-bench Lite",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 314,
          "metricId": "resolved",
          "observedAt": "2025-02-07",
          "protocol": {
            "checked": false,
            "harness": "Aegis",
            "leaderboard_variant": "Lite",
            "scaffold": "Aegis",
            "subject_type": "system"
          },
          "rawValue": 30.33,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[4]=Lite;results[49]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 30.33
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "o3-mini_1.0",
      "numericRowCount": 1,
      "observedAtMax": "2025-02-07",
      "observedAtMin": "2025-02-07",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    },
    {
      "benchmarkCount": 1,
      "benchmarkIds": [
        "swebench-verified"
      ],
      "examples": [
        {
          "artifact": "swebench-official/candidates.jsonl",
          "benchmarkId": "swebench-verified",
          "benchmarkName": "SWE-bench Verified",
          "evidenceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "line": 157,
          "metricId": "resolved",
          "observedAt": "2024-12-21",
          "protocol": {
            "checked": false,
            "harness": "CodeStory Midwit Agent",
            "leaderboard_variant": "Verified",
            "scaffold": "CodeStory Midwit Agent",
            "subject_type": "system"
          },
          "rawValue": 62.2,
          "sourceId": "swebench-official",
          "sourceLabel": "SWE-bench · official leaderboard JSON",
          "sourceLocator": "leaderboards[3]=Verified;results[72]",
          "sourceUrl": "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json",
          "unit": "percent",
          "value": 62.2
        }
      ],
      "latestRetrievedAt": "2026-08-27T04:57:09Z",
      "mappingCandidates": [],
      "mappingStatusCounts": {
        "unmatched": 1
      },
      "metricIds": [
        "resolved"
      ],
      "modelRef": "swe-search",
      "numericRowCount": 1,
      "observedAtMax": "2024-12-21",
      "observedAtMin": "2024-12-21",
      "rowCount": 1,
      "sourceId": "swebench-official",
      "sourceIds": [
        "swebench-official"
      ],
      "sourceLabels": [
        "SWE-bench · official leaderboard JSON"
      ],
      "sourceUrls": [
        "https://raw.githubusercontent.com/swe-bench/swe-bench.github.io/master/data/leaderboards.json"
      ],
      "statusCounts": {
        "reported": 1
      }
    }
  ],
  "meta": {
    "fullRowsPath": "data/public/unmapped.jsonl",
    "generatedAt": "2026-08-27T16:21:44Z",
    "note": "Source model spellings without a safe canonical release alias; inspect and promote only with an explicit, source-scoped review.",
    "publicEvidenceGeneratedAt": "2026-08-27T16:21:44Z",
    "schemaVersion": "public-unmapped-summary@0.1",
    "status": "unmapped-unreviewed",
    "verified": false
  },
  "stats": {
    "benchmarkCounts": {
      "agents-last-exam": 144,
      "aider-polyglot": 123,
      "aime-2026": 16,
      "arena-agent": 13,
      "arena-document": 6,
      "arena-search": 15,
      "arena-text": 721,
      "arena-vision": 561,
      "arena-webdev": 137,
      "bfcl": 93,
      "epoch-adversarial_nli_external": 15,
      "epoch-ale_bench_external": 35,
      "epoch-algotune_external": 7,
      "epoch-apex_agents_external": 16,
      "epoch-arc_agi_2_external": 65,
      "epoch-arc_agi_external": 73,
      "epoch-arc_ai2_external": 134,
      "epoch-balrog_external": 27,
      "epoch-bbh_external": 88,
      "epoch-blueprint_bench_2_external": 4,
      "epoch-bool_q_external": 136,
      "epoch-chess_puzzles": 41,
      "epoch-cl_bench_external": 6,
      "epoch-cl_bench_life_external": 1,
      "epoch-common_sense_qa_2_external": 6,
      "epoch-critpt_external": 58,
      "epoch-cursorbench_external": 1,
      "epoch-deepresearchbench_external": 12,
      "epoch-deepswe_external": 2,
      "epoch-enigma_eval_external": 28,
      "epoch-epoch_capabilities_index": 290,
      "epoch-forecastbench_external": 51,
      "epoch-frontiercode_external": 5,
      "epoch-frontiermath_tier_4": 25,
      "epoch-frontierswe_external": 1,
      "epoch-gbaeval_external": 2,
      "epoch-gdp_pdf_external": 7,
      "epoch-gdpval_external": 6,
      "epoch-geobench_external": 29,
      "epoch-gsm8k_external": 170,
      "epoch-gso_external": 15,
      "epoch-hella_swag_external": 113,
      "epoch-lambada_external": 53,
      "epoch-lech_mazur_writing_external": 43,
      "epoch-math_level_5": 97,
      "epoch-metr_time_horizons_external": 35,
      "epoch-mindcube_external": 5,
      "epoch-mmlu_external": 217,
      "epoch-mystery_game_puzzles": 5,
      "epoch-open_book_qa_external": 70,
      "epoch-otis_mock_aime_2024_2025": 113,
      "epoch-piqa_external": 113,
      "epoch-proofbench_external": 14,
      "epoch-rli_external": 2,
      "epoch-scicode_external": 51,
      "epoch-science_qa_external": 26,
      "epoch-simplebench_external": 52,
      "epoch-spatialviz_bench_external": 6,
      "epoch-superglue_external": 10,
      "epoch-surface_evolver_bench_external": 6,
      "epoch-terminalbench_external": 59,
      "epoch-the_agent_company_external": 15,
      "epoch-trivia_qa_external": 115,
      "epoch-vending_bench_2_external": 11,
      "epoch-vpct_external": 23,
      "epoch-webdev_arena_external": 26,
      "epoch-weirdml_external": 86,
      "epoch-wino_grande_external": 137,
      "frontiermath": 51,
      "gpqa-diamond": 225,
      "helm-gpqa": 413,
      "helm-gsm8k": 637,
      "helm-ifeval": 413,
      "helm-legalbench": 637,
      "helm-math": 637,
      "helm-mean-score": 118,
      "helm-mean-win-rate": 273,
      "helm-medqa": 637,
      "helm-mmlu": 637,
      "helm-mmlu-pro": 413,
      "helm-narrativeqa": 637,
      "helm-naturalquestions-closed-book": 637,
      "helm-naturalquestions-open-book": 637,
      "helm-omni-math": 413,
      "helm-openbookqa": 637,
      "helm-wildbench": 413,
      "helm-wmt-2014": 637,
      "hle": 90,
      "hmmt": 9,
      "livebench": 53,
      "livebench-amps-hard": 7,
      "livebench-code-completion": 7,
      "livebench-code-generation": 7,
      "livebench-connections": 7,
      "livebench-consecutive-events": 7,
      "livebench-integrals-with-game": 7,
      "livebench-javascript": 7,
      "livebench-logic-with-navigation": 7,
      "livebench-math-comp": 7,
      "livebench-olympiad": 7,
      "livebench-paraphrase": 7,
      "livebench-plot-unscrambling": 7,
      "livebench-python": 7,
      "livebench-simplify": 7,
      "livebench-spatial": 7,
      "livebench-story-generation": 7,
      "livebench-summarize": 7,
      "livebench-tablejoin": 7,
      "livebench-tablereformat": 7,
      "livebench-theory-of-mind": 7,
      "livebench-typescript": 7,
      "livebench-typos": 7,
      "livebench-zebra-puzzle": 7,
      "mle-bench": 84,
      "mmlu-pro": 130,
      "osworld": 14,
      "simpleqa": 22,
      "swebench-bash-only": 22,
      "swebench-lite": 84,
      "swebench-multilingual": 2,
      "swebench-multimodal": 22,
      "swebench-pro": 24,
      "swebench-test": 23,
      "swebench-verified": 215,
      "terminal-bench": 21
    },
    "mappingCounts": {
      "unmatched": 14766
    },
    "numericRows": 14668,
    "publicDeduplicatedRows": 21227,
    "rows": 14766,
    "sourceCounts": {
      "agents-last-exam": 144,
      "helm-capabilities": 2183,
      "helm-lite": 6643,
      "hf-aime-2026": 16,
      "hf-gpqa": 86,
      "hf-hle": 58,
      "hf-hmmt-2026": 9,
      "hf-mmlu-pro": 130,
      "hf-swebench-pro": 24,
      "hf-swebench-verified": 59,
      "hf-terminal-bench": 21,
      "livebench-official": 161,
      "lmarena-hf-dataset": 1453,
      "src-aider-polyglot": 60,
      "src-bfcl": 93,
      "src-epoch-benchmark-hub": 3242,
      "src-mle-bench": 84,
      "swebench-official": 300
    },
    "statusCounts": {
      "candidate": 98,
      "reported": 14668
    },
    "uniqueAliases": 1591,
    "uniqueBenchmarks": 125,
    "uniqueSources": 18
  }
}
