{
  "schemaVersion": 1,
  "asOfDate": "2026-09-07",
  "reviewedAt": "2026-09-07",
  "categoryCount": 6,
  "staleCategoryCount": 0,
  "reviewSoonCategoryCount": 0,
  "olderSourceSnapshotCount": 3,
  "nextReviewDueDate": "2026-09-21",
  "policy": {
    "rule": "A benchmark category becomes build-blocking when its explicit verification date exceeds its category-specific review cadence.",
    "sourceAgeRule": "An old official source snapshot is disclosed separately from review freshness. A recently rechecked static benchmark may remain usable, but the page must show that its model pool has not been refreshed.",
    "coverageRule": "Selected feeds and deduplicated system lists must never be described as complete global leaderboards.",
    "comparisonRule": "Agent-and-model system results must not be presented as model-only scores."
  },
  "categories": [
    {
      "id": "general-reasoning",
      "category": "General reasoning",
      "benchmarkId": "general365",
      "benchmarkName": "General365",
      "benchmarkVersion": "Original 2026 release",
      "benchmarkPeriod": "2026 benchmark",
      "benchmarkReleaseDate": "2026-04-13",
      "sourceSnapshotDate": "2026-04-13",
      "sourceRevision": "arXiv:2604.11778 and public leaderboard",
      "verifiedAt": "2026-09-07",
      "reviewCadenceDays": 45,
      "reviewDueDate": "2026-10-22",
      "reviewStatus": "current",
      "reviewAgeDays": 0,
      "sourceSnapshotAgeDays": 147,
      "sourceSnapshotStatus": "older-source-snapshot",
      "scope": "model-only",
      "scopeLabel": "Model-only evaluation",
      "coverageMode": "official-published-top-set",
      "coverageLabel": "Official published top set",
      "freshnessNote": "The official paper and project leaderboard were rechecked on July 25, 2026. The benchmark snapshot remains the original April 2026 release.",
      "modelWatch": [],
      "primaryScoreKey": "score",
      "primaryScoreLabel": "Overall accuracy",
      "latestResultDate": "2026-04-13",
      "currentYearResultCount": 10,
      "sources": [
        {
          "title": "General365 official leaderboard",
          "publisher": "Meituan LongCat Team",
          "url": "https://general365.github.io/",
          "type": "Official 2026 general-reasoning benchmark leaderboard",
          "id": "general-reasoning-source-1",
          "role": "Official leaderboard or maintained feed",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-04-13",
          "revision": "arXiv:2604.11778 and public leaderboard"
        },
        {
          "title": "General365 research paper",
          "publisher": "arXiv",
          "url": "https://arxiv.org/abs/2604.11778",
          "type": "Primary research paper",
          "id": "general-reasoning-source-2",
          "role": "Primary methodology or research source",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-04-13",
          "revision": "arXiv:2604.11778 and public leaderboard"
        }
      ],
      "items": [
        {
          "model": "Gemini 3.8 Pro",
          "provider": "Google",
          "origin": "International",
          "score": 64.2,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:gemini-3-8-pro:1",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "ChatGPT Astra Thinking",
          "provider": "OpenAI",
          "origin": "International",
          "score": 63.1,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:chatgpt-astra-thinking:2",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Gemini 3 Pro",
          "provider": "Google",
          "origin": "International",
          "score": 62.8,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:gemini-3-pro:3",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Gemini 3.8 Flash",
          "provider": "Google",
          "origin": "International",
          "score": 62.4,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:gemini-3-8-flash:4",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Gemini 3 Flash",
          "provider": "Google",
          "origin": "International",
          "score": 60.8,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:gemini-3-flash:5",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "GLM-5 Thinking",
          "provider": "Z.ai",
          "origin": "China",
          "score": 59.9,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:glm-5-thinking:6",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "GPT-5 Thinking",
          "provider": "OpenAI",
          "origin": "International",
          "score": 58.6,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:gpt-5-thinking:7",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "GPT-5.1 Thinking",
          "provider": "OpenAI",
          "origin": "International",
          "score": 58.2,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:gpt-5-1-thinking:8",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3.5-397B-A17B Thinking",
          "provider": "Qwen",
          "origin": "China",
          "score": 57.7,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:qwen3-5-397b-a17b-thinking:9",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "DeepSeek-V3.2-Speciale",
          "provider": "DeepSeek",
          "origin": "China",
          "score": 57.5,
          "resultDate": "2026-04-13",
          "exactSystemId": "general-reasoning:deepseek-v3-2-speciale:10",
          "resultSnapshotDate": "2026-04-13",
          "evaluationMode": "Published General365 evaluation",
          "systemScope": "model-only",
          "sourceRowUrl": null
        }
      ]
    },
    {
      "id": "mathematics",
      "category": "Mathematics",
      "benchmarkId": "aime-hmmt-2026-openevals",
      "benchmarkName": "AIME 2026 + HMMT 2026",
      "benchmarkVersion": "OpenEvals selected-feed snapshot",
      "benchmarkPeriod": "2026 examination cycle",
      "benchmarkReleaseDate": null,
      "sourceSnapshotDate": "2026-07-24",
      "sourceRevision": "OpenEvals leaderboard-data snapshot used by the July 24 intelligence registry",
      "verifiedAt": "2026-09-07",
      "reviewCadenceDays": 30,
      "reviewDueDate": "2026-10-07",
      "reviewStatus": "current",
      "reviewAgeDays": 0,
      "sourceSnapshotAgeDays": 45,
      "sourceSnapshotStatus": "recent-source-snapshot",
      "scope": "model-only",
      "scopeLabel": "Model-only reported results",
      "coverageMode": "selected-official-feed-rows",
      "coverageLabel": "Selected tracked systems, not the complete feed",
      "freshnessNote": "The displayed rows are a selected tracked-system snapshot from the maintained OpenEvals feed. They are not presented as a complete global leaderboard.",
      "modelWatch": [],
      "primaryScoreKey": "mean",
      "primaryScoreLabel": "Category mean",
      "latestResultDate": "2026-07-24",
      "currentYearResultCount": 8,
      "sources": [
        {
          "title": "OpenEvals leaderboard-data",
          "publisher": "Hugging Face OpenEvals",
          "url": "https://huggingface.co/datasets/OpenEvals/leaderboard-data",
          "type": "Official multi-benchmark data aggregation",
          "id": "mathematics-source-1",
          "role": "Official leaderboard or maintained feed",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-07-24",
          "revision": "OpenEvals leaderboard-data snapshot used by the July 24 intelligence registry"
        }
      ],
      "items": [
        {
          "model": "Kimi K2.5",
          "provider": "Moonshot AI",
          "origin": "China",
          "aime2026": 95.83,
          "hmmt2026": 87.12,
          "mean": 91.5,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:kimi-k2-5:1",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Step-3.5-Flash",
          "provider": "StepFun",
          "origin": "China",
          "aime2026": 96.67,
          "hmmt2026": 86.36,
          "mean": 91.5,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:step-3-5-flash:2",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "GLM-5",
          "provider": "Z.ai",
          "origin": "China",
          "aime2026": 95.83,
          "hmmt2026": 86.36,
          "mean": 91.1,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:glm-5:3",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3.5-397B-A17B",
          "provider": "Qwen",
          "origin": "China",
          "aime2026": 93.33,
          "hmmt2026": 87.88,
          "mean": 90.6,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:qwen3-5-397b-a17b:4",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "DeepSeek-V3.2",
          "provider": "DeepSeek",
          "origin": "China",
          "aime2026": 94.17,
          "hmmt2026": 84.09,
          "mean": 89.1,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:deepseek-v3-2:5",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3.5-35B-A3B",
          "provider": "Qwen",
          "origin": "China",
          "aime2026": 93.33,
          "hmmt2026": 81.82,
          "mean": 87.6,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:qwen3-5-35b-a3b:6",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "NVIDIA Nemotron 3 Super 120B-A12B",
          "provider": "NVIDIA",
          "origin": "International",
          "aime2026": 90,
          "hmmt2026": 84.85,
          "mean": 87.4,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:nvidia-nemotron-3-super-120b-a12b:7",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3.5-27B",
          "provider": "Qwen",
          "origin": "China",
          "aime2026": 90.83,
          "hmmt2026": 81.06,
          "mean": 85.9,
          "resultDate": "2026-07-24",
          "exactSystemId": "mathematics:qwen3-5-27b:8",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "OpenEvals official-feed percentages",
          "systemScope": "model-only",
          "sourceRowUrl": null
        }
      ]
    },
    {
      "id": "software-engineering",
      "category": "Software engineering",
      "benchmarkId": "swe-bench-verified",
      "benchmarkName": "SWE-bench Verified",
      "benchmarkVersion": "500 human-validated issues",
      "benchmarkPeriod": "Maintained benchmark",
      "benchmarkReleaseDate": null,
      "sourceSnapshotDate": "2026-07-24",
      "sourceRevision": "Selected official-feed result snapshot",
      "verifiedAt": "2026-09-07",
      "reviewCadenceDays": 30,
      "reviewDueDate": "2026-10-07",
      "reviewStatus": "current",
      "reviewAgeDays": 0,
      "sourceSnapshotAgeDays": 45,
      "sourceSnapshotStatus": "recent-source-snapshot",
      "scope": "agent-and-model-system",
      "scopeLabel": "System result; harness and settings matter",
      "coverageMode": "selected-official-feed-rows",
      "coverageLabel": "Selected tracked systems, not the complete leaderboard",
      "freshnessNote": "The official maintained benchmark was rechecked on July 25, 2026. The displayed values remain selected tracked-system rows rather than a claim to reproduce the full live leaderboard.",
      "modelWatch": [],
      "primaryScoreKey": "score",
      "primaryScoreLabel": "Issues resolved",
      "latestResultDate": "2026-07-24",
      "currentYearResultCount": 8,
      "sources": [
        {
          "title": "SWE-bench Verified benchmark",
          "publisher": "SWE-bench",
          "url": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Verified",
          "id": "software-engineering-source-1",
          "role": "Official leaderboard or maintained feed",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-07-24",
          "revision": "Selected official-feed result snapshot",
          "type": "Official benchmark source"
        }
      ],
      "items": [
        {
          "model": "Qwen3.5-397B-A17B",
          "provider": "Qwen",
          "origin": "China",
          "score": 76.4,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:qwen3-5-397b-a17b:1",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "Step-3.5-Flash",
          "provider": "StepFun",
          "origin": "China",
          "score": 74.4,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:step-3-5-flash:2",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "GLM-5",
          "provider": "Z.ai",
          "origin": "China",
          "score": 72.8,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:glm-5:3",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3.5-27B",
          "provider": "Qwen",
          "origin": "China",
          "score": 72.4,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:qwen3-5-27b:4",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "Kimi K2.5",
          "provider": "Moonshot AI",
          "origin": "China",
          "score": 70.8,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:kimi-k2-5:5",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "DeepSeek-V3.2",
          "provider": "DeepSeek",
          "origin": "China",
          "score": 70,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:deepseek-v3-2:6",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3.5-35B-A3B",
          "provider": "Qwen",
          "origin": "China",
          "score": 69.2,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:qwen3-5-35b-a3b:7",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        },
        {
          "model": "NVIDIA Nemotron 3 Super 120B-A12B",
          "provider": "NVIDIA",
          "origin": "International",
          "score": 53.73,
          "resultDate": "2026-07-24",
          "exactSystemId": "software-engineering:nvidia-nemotron-3-super-120b-a12b:8",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Official feed system result; exact harness may vary",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": null
        }
      ]
    },
    {
      "id": "terminal-agents",
      "category": "Terminal agents",
      "benchmarkId": "terminal-bench-2-1",
      "benchmarkName": "Terminal-Bench 2.1",
      "benchmarkVersion": "terminal-bench@2.1",
      "benchmarkPeriod": "Live maintained leaderboard",
      "benchmarkReleaseDate": "2026-05-06",
      "sourceSnapshotDate": "2026-07-11",
      "sourceRevision": "Official leaderboard with benchmark-team-verified submissions",
      "verifiedAt": "2026-09-07",
      "reviewCadenceDays": 14,
      "reviewDueDate": "2026-09-21",
      "reviewStatus": "current",
      "reviewAgeDays": 0,
      "sourceSnapshotAgeDays": 58,
      "sourceSnapshotStatus": "older-source-snapshot",
      "scope": "agent-and-model-system",
      "scopeLabel": "Agent + model system",
      "coverageMode": "deduplicated-official-model-selection",
      "coverageLabel": "One official submission per model",
      "freshnessNote": "The official 17-entry Terminal-Bench 2.1 leaderboard was rechecked on July 25, 2026. The page keeps one official submission per model so repeated harness variants do not dominate the display.",
      "modelWatch": [
        "Kimi K3",
        "GLM-5.2",
        "Qwen3.7 Max",
        "MiniMax-M3",
        "DeepSeek V4 Pro"
      ],
      "primaryScoreKey": "score",
      "primaryScoreLabel": "Accuracy",
      "latestResultDate": "2026-07-11",
      "currentYearResultCount": 8,
      "sources": [
        {
          "id": "terminal-bench-21-leaderboard",
          "title": "terminal-bench@2.1 leaderboard",
          "publisher": "Terminal-Bench / Harbor Framework",
          "url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
          "published": "Live maintained leaderboard",
          "retrieved": "2026-07-24",
          "type": "Official benchmark leaderboard",
          "role": "Official leaderboard or maintained feed",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-07-11",
          "revision": "Official leaderboard with benchmark-team-verified submissions"
        },
        {
          "id": "terminal-bench-21-release",
          "title": "Terminal-Bench 2.1 release and methodology notes",
          "publisher": "Terminal-Bench",
          "url": "https://www.tbench.ai/news/terminal-bench-2-1",
          "published": "2026-05-06",
          "retrieved": "2026-07-24",
          "type": "Official benchmark documentation",
          "role": "Primary methodology or research source",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-07-11",
          "revision": "Official leaderboard with benchmark-team-verified submissions"
        }
      ],
      "items": [
        {
          "displayLabel": "Fable 5 + Claude Code",
          "model": "Fable 5",
          "provider": "Anthropic",
          "origin": "International",
          "agent": "Claude Code",
          "agentOrg": "Anthropic",
          "effort": "xhigh",
          "score": 83.8,
          "uncertainty": "± 1.2%",
          "date": "2026-06-07",
          "cost": "$552.67",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/75",
          "exactSystemId": "terminal-agents:fable-5-claude-code:1",
          "resultSnapshotDate": "2026-06-07",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/75"
        },
        {
          "displayLabel": "GPT-5.5 + Codex",
          "model": "GPT-5.5",
          "provider": "OpenAI",
          "origin": "International",
          "agent": "Codex",
          "agentOrg": "OpenAI",
          "effort": "xhigh",
          "score": 83.1,
          "uncertainty": "± 1.1%",
          "date": "2026-05-01",
          "cost": "$2,059.19",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/45",
          "exactSystemId": "terminal-agents:gpt-5-5-codex:2",
          "resultSnapshotDate": "2026-05-01",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/45"
        },
        {
          "displayLabel": "Grok 4.5 + Cursor CLI",
          "model": "Grok 4.5",
          "provider": "xAI",
          "origin": "International",
          "agent": "Cursor CLI",
          "agentOrg": "Cursor",
          "effort": "high",
          "score": 79.3,
          "uncertainty": "± 1.5%",
          "date": "2026-07-09",
          "cost": "$134.09",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/86",
          "exactSystemId": "terminal-agents:grok-4-5-cursor-cli:3",
          "resultSnapshotDate": "2026-07-09",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/86"
        },
        {
          "displayLabel": "Opus 4.8 + Claude Code",
          "model": "Claude Opus 4.8",
          "provider": "Anthropic",
          "origin": "International",
          "agent": "Claude Code",
          "agentOrg": "Anthropic",
          "effort": "high",
          "score": 78.9,
          "uncertainty": "± 1.3%",
          "date": "2026-07-09",
          "cost": "$286.94",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/92",
          "exactSystemId": "terminal-agents:opus-4-8-claude-code:4",
          "resultSnapshotDate": "2026-07-09",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/92"
        },
        {
          "displayLabel": "GPT-5.6 Terra + Codex",
          "model": "GPT-5.6 Terra",
          "provider": "OpenAI",
          "origin": "International",
          "agent": "Codex",
          "agentOrg": "OpenAI",
          "effort": "max",
          "score": 78.4,
          "uncertainty": "± 1.3%",
          "date": "2026-07-11",
          "cost": "$421.15",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/115",
          "exactSystemId": "terminal-agents:gpt-5-6-terra-codex:5",
          "resultSnapshotDate": "2026-07-11",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/115"
        },
        {
          "displayLabel": "Muse Spark 1.1 + mini-SWE-agent",
          "model": "Muse Spark 1.1",
          "provider": "Meta",
          "origin": "International",
          "agent": "mini-SWE-agent",
          "agentOrg": "Princeton",
          "effort": "xhigh",
          "score": 76.2,
          "uncertainty": "± 1.2%",
          "date": "2026-07-09",
          "cost": "$198.05",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/94",
          "exactSystemId": "terminal-agents:muse-spark-1-1-mini-swe-agent:6",
          "resultSnapshotDate": "2026-07-09",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/94"
        },
        {
          "displayLabel": "Gemini 3 Pro + Terminus 2",
          "model": "Gemini 3 Pro",
          "provider": "Google",
          "origin": "International",
          "agent": "Terminus 2",
          "agentOrg": "Terminal-Bench",
          "effort": "high",
          "score": 73.9,
          "uncertainty": "± 1.3%",
          "date": "2026-05-01",
          "cost": "$224.44",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/48",
          "exactSystemId": "terminal-agents:gemini-3-pro-terminus-2:7",
          "resultSnapshotDate": "2026-05-01",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/48"
        },
        {
          "displayLabel": "GLM-5.1 + Claude Code",
          "model": "GLM-5.1",
          "provider": "Z.ai",
          "origin": "China",
          "agent": "Claude Code",
          "agentOrg": "Anthropic",
          "effort": "max",
          "score": 58.7,
          "uncertainty": "± 1.2%",
          "date": "2026-05-01",
          "cost": "$277.14",
          "submissionUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/67",
          "exactSystemId": "terminal-agents:glm-5-1-claude-code:8",
          "resultSnapshotDate": "2026-05-01",
          "evaluationMode": "Official agent-and-model submission",
          "systemScope": "agent-and-model-system",
          "sourceRowUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/67"
        }
      ]
    },
    {
      "id": "long-context",
      "category": "Long context",
      "benchmarkId": "longbench-pro",
      "benchmarkName": "LongBench Pro",
      "benchmarkVersion": "arXiv:2601.02872 / official Space output",
      "benchmarkPeriod": "2026 benchmark",
      "benchmarkReleaseDate": "2026-01-06",
      "sourceSnapshotDate": "2026-01-08",
      "sourceRevision": "Hugging Face Space output commit 5e81af2; app commit 05c3037",
      "verifiedAt": "2026-09-07",
      "reviewCadenceDays": 30,
      "reviewDueDate": "2026-10-07",
      "reviewStatus": "current",
      "reviewAgeDays": 0,
      "sourceSnapshotAgeDays": 242,
      "sourceSnapshotStatus": "older-source-snapshot",
      "scope": "model-only",
      "scopeLabel": "Model-only benchmark",
      "coverageMode": "official-published-top-set",
      "coverageLabel": "Official published snapshot",
      "freshnessNote": "The official source was rechecked on July 25, 2026, but its published result files remain an older January 2026 snapshot. The page therefore labels the source age explicitly instead of presenting the model list as newly updated.",
      "modelWatch": [
        "Gemini 3 family",
        "Claude 5 family",
        "GPT-5.5 and GPT-5.6",
        "Qwen3.5 family",
        "Kimi K2.5"
      ],
      "primaryScoreKey": "score",
      "primaryScoreLabel": "Overall score",
      "latestResultDate": "2026-01-08",
      "currentYearResultCount": 10,
      "sources": [
        {
          "title": "LongBench Pro official leaderboard",
          "publisher": "KCSG, IIE, Chinese Academy of Sciences",
          "url": "https://huggingface.co/spaces/caskcsg/LongBench-Pro-Leaderboard",
          "type": "Official 2026 long-context benchmark leaderboard",
          "id": "long-context-source-1",
          "role": "Official leaderboard or maintained feed",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-01-08",
          "revision": "Hugging Face Space output commit 5e81af2; app commit 05c3037"
        },
        {
          "title": "LongBench Pro research paper",
          "publisher": "arXiv",
          "url": "https://arxiv.org/abs/2601.02872",
          "type": "Primary research paper",
          "id": "long-context-source-2",
          "role": "Primary methodology or research source",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-01-08",
          "revision": "Hugging Face Space output commit 5e81af2; app commit 05c3037"
        }
      ],
      "items": [
        {
          "model": "Gemini 3.8 Pro",
          "provider": "Google",
          "origin": "International",
          "score": 77.2,
          "mode": "Thinking",
          "contextCap": "1M",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:gemini-3-8-pro:1",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "ChatGPT Astra",
          "provider": "OpenAI",
          "origin": "International",
          "score": 74.8,
          "mode": "Thinking",
          "contextCap": "512K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:chatgpt-astra:2",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Gemini 2.5 Pro",
          "provider": "Google",
          "origin": "International",
          "score": 73.4,
          "mode": "Thinking",
          "contextCap": "1M",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:gemini-2-5-pro:3",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "GPT-5",
          "provider": "OpenAI",
          "origin": "International",
          "score": 72.6,
          "mode": "Thinking",
          "contextCap": "272K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:gpt-5:4",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Claude 4 Sonnet",
          "provider": "Anthropic",
          "origin": "International",
          "score": 69.9,
          "mode": "Thinking",
          "contextCap": "1M",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:claude-4-sonnet:5",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "DeepSeek-V3.2",
          "provider": "DeepSeek",
          "origin": "China",
          "score": 67.8,
          "mode": "Thinking",
          "contextCap": "120K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:deepseek-v3-2:6",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3-235B-A22B Thinking",
          "provider": "Qwen",
          "origin": "China",
          "score": 67,
          "mode": "Thinking",
          "contextCap": "224K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:qwen3-235b-a22b-thinking:7",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3-Next-80B-A3B Thinking",
          "provider": "Qwen",
          "origin": "China",
          "score": 64,
          "mode": "Thinking",
          "contextCap": "224K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking",
          "exactSystemId": "long-context:qwen3-next-80b-a3b-thinking:8",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Qwen3-Next-80B-A3B Instruct",
          "provider": "Qwen",
          "origin": "China",
          "score": 60.8,
          "mode": "Thinking prompt",
          "contextCap": "224K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking prompt",
          "exactSystemId": "long-context:qwen3-next-80b-a3b-instruct:9",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        },
        {
          "model": "Kimi K2 Instruct 0905",
          "provider": "Moonshot AI",
          "origin": "China",
          "score": 55.5,
          "mode": "Thinking prompt",
          "contextCap": "224K",
          "resultDate": "2026-01-08",
          "evaluationMode": "Thinking prompt",
          "exactSystemId": "long-context:kimi-k2-instruct-0905:10",
          "resultSnapshotDate": "2026-01-08",
          "systemScope": "model-only",
          "sourceRowUrl": null
        }
      ]
    },
    {
      "id": "document-ocr",
      "category": "Document and OCR",
      "benchmarkId": "olmocr-openevals",
      "benchmarkName": "olmOCR",
      "benchmarkVersion": "OpenEvals selected specialist snapshot",
      "benchmarkPeriod": "Maintained specialist benchmark feed",
      "benchmarkReleaseDate": null,
      "sourceSnapshotDate": "2026-07-24",
      "sourceRevision": "OpenEvals leaderboard-data snapshot used by the July 24 intelligence registry",
      "verifiedAt": "2026-09-07",
      "reviewCadenceDays": 30,
      "reviewDueDate": "2026-10-07",
      "reviewStatus": "current",
      "reviewAgeDays": 0,
      "sourceSnapshotAgeDays": 45,
      "sourceSnapshotStatus": "recent-source-snapshot",
      "scope": "specialist-document-system",
      "scopeLabel": "Specialist document/OCR systems",
      "coverageMode": "selected-official-feed-rows",
      "coverageLabel": "Selected specialist systems",
      "freshnessNote": "The displayed specialist rows were rechecked against the maintained OpenEvals source on July 25, 2026. Coverage remains intentionally limited to directly comparable document/OCR systems.",
      "modelWatch": [],
      "primaryScoreKey": "score",
      "primaryScoreLabel": "Document score",
      "latestResultDate": "2026-07-24",
      "currentYearResultCount": 10,
      "sources": [
        {
          "title": "OpenEvals leaderboard-data",
          "publisher": "Hugging Face OpenEvals",
          "url": "https://huggingface.co/datasets/OpenEvals/leaderboard-data",
          "type": "Official multi-benchmark data aggregation",
          "id": "document-ocr-source-1",
          "role": "Official leaderboard or maintained feed",
          "checkedAt": "2026-09-07",
          "snapshotDate": "2026-07-24",
          "revision": "OpenEvals leaderboard-data snapshot used by the July 24 intelligence registry"
        }
      ],
      "items": [
        {
          "model": "chandra-ocr-2",
          "provider": "Datalab",
          "origin": "International",
          "score": 85.9,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:chandra-ocr-2:1",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "dots.mocr",
          "provider": "RedNote HiLab",
          "origin": "China",
          "score": 83.9,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:dots-mocr:2",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "LightOnOCR-2-1B",
          "provider": "LightOn AI",
          "origin": "International",
          "score": 83.2,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:lightonocr-2-1b:3",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "chandra",
          "provider": "Datalab",
          "origin": "International",
          "score": 83.1,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:chandra:4",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "Infinity-Parser-7B",
          "provider": "Infiny AI",
          "origin": "International",
          "score": 82.5,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:infinity-parser-7b:5",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "olmOCR-2-7B FP8",
          "provider": "AllenAI",
          "origin": "International",
          "score": 82.4,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:olmocr-2-7b-fp8:6",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "PaddleOCR-VL",
          "provider": "PaddlePaddle",
          "origin": "China",
          "score": 80,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:paddleocr-vl:7",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "Qianfan-OCR",
          "provider": "Baidu",
          "origin": "China",
          "score": 79.8,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:qianfan-ocr:8",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "DeepSeek-OCR-2",
          "provider": "DeepSeek",
          "origin": "China",
          "score": 76.3,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:deepseek-ocr-2:9",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        },
        {
          "model": "GLM-OCR",
          "provider": "Z.ai",
          "origin": "China",
          "score": 75.2,
          "resultDate": "2026-07-24",
          "exactSystemId": "document-ocr:glm-ocr:10",
          "resultSnapshotDate": "2026-07-24",
          "evaluationMode": "Maintained specialist benchmark feed",
          "systemScope": "specialist-document-system",
          "sourceRowUrl": null
        }
      ]
    }
  ]
}
