{
  "as_of": "2026-09-25",
  "generated_at": "2026-09-25 07:00 CEST",
  "weight_scheme": "Agent/system 70% · Reasoning 20% · Chat 10%. AA · DeepSWE 20% each; TB · BrowseComp · HLE 15% each.",
  "weight_scheme_full": "Hermes buckets: agent/system 70% (Terminal-Bench 4.0 @15% multi-step tool/terminal (BenchLM mirror of tbench.ai; best harness/effort per model; not comparable to TB 2.x), DeepSWE 20% code exec, Agent Arena 20% rank percentile, BrowseComp 15% web research); reasoning 20% (HLE 15%, GPQA 5%); chat 10% (LMSYS). SWE-Atlas-QnA retired (not Hermes task mix; no decoration weights). BFCL-class / tau-bench evaluated 2026-08-30 and absent (floor fail). Sources covering <80% of the scored set are weight-capped at 15% then renormalized. Primary missing→z=0; sensitivity=renorm. Z=median/MAD with ε floor and ±3 clamp. Hermes Index is an index, not a percentage.",
  "weights": [
    {
      "label": "Terminal-Bench 4.0",
      "key": "terminalbench",
      "weight": 0.15,
      "active_weight": 0.1667,
      "effective_weight": 0.1138,
      "bucket": "Agent/system",
      "counts_as_direct": true
    },
    {
      "label": "DeepSWE",
      "key": "deepswe",
      "weight": 0.2,
      "active_weight": 0.1667,
      "effective_weight": 0.1743,
      "bucket": "Agent/system",
      "counts_as_direct": true
    },
    {
      "label": "Agent Arena",
      "key": "agent_lmsys",
      "weight": 0.2,
      "active_weight": 0.1667,
      "effective_weight": 0.1135,
      "bucket": "Agent/system",
      "counts_as_direct": false
    },
    {
      "label": "BrowseComp",
      "key": "browsecomp",
      "weight": 0.15,
      "active_weight": 0.1667,
      "effective_weight": 0.2015,
      "bucket": "Agent/system",
      "counts_as_direct": true
    },
    {
      "label": "HLE",
      "key": "hle",
      "weight": 0.15,
      "active_weight": 0.1667,
      "effective_weight": 0.2693,
      "bucket": "Reasoning",
      "counts_as_direct": true
    },
    {
      "label": "LMSYS Text Arena",
      "key": "lmsys",
      "weight": 0.1,
      "active_weight": 0.1111,
      "effective_weight": 0.0991,
      "bucket": "Chat",
      "counts_as_direct": true
    },
    {
      "label": "GPQA Diamond",
      "key": "gpqa_diamond",
      "weight": 0.05,
      "active_weight": 0.0556,
      "effective_weight": 0.0285,
      "bucket": "Reasoning",
      "counts_as_direct": true
    }
  ],
  "scoring": {
    "active_weights": {
      "terminalbench": 0.1667,
      "deepswe": 0.1667,
      "browsecomp": 0.1667,
      "agent_lmsys": 0.1667,
      "hle": 0.1667,
      "lmsys": 0.1111,
      "gpqa_diamond": 0.0556
    },
    "effective_weights": {
      "terminalbench": 0.1138,
      "deepswe": 0.1743,
      "browsecomp": 0.2015,
      "agent_lmsys": 0.1135,
      "hle": 0.2693,
      "lmsys": 0.0991,
      "gpqa_diamond": 0.0285
    },
    "weight_caps_hit": {
      "deepswe": {
        "nominal": 0.2,
        "capped": 0.15,
        "coverage_frac": 0.735
      },
      "agent_lmsys": {
        "nominal": 0.2,
        "capped": 0.15,
        "coverage_frac": 0.676
      }
    },
    "mad_eps": {
      "terminalbench": 1.0,
      "deepswe": 1.0,
      "browsecomp": 1.0,
      "agent_lmsys": 1.0,
      "hle": 1.0,
      "lmsys": 10.0,
      "gpqa_diamond": 1.0
    },
    "z_clamp": 3.0,
    "list_price_def": "input+output USD per 1M tokens"
  },
  "composite_formula": "Each active source contributes one metric. Percentage benches that arrive as 0–1 fractions are scaled to percent; there is no fixed bound map before ranking. For every source, we compute a robust center and scale on the fixed scored-model set (models with at least four active direct scores): the center is the median, and the scale is max(1.4826×MAD, ε), with ε = 1.0 percentage point for percent benches and ε = 10 Elo for LMSYS. Each model’s per-source z is (metric − center) / scale, then clamped to ±3. The Hermes Index is the weighted mean of those z values under the active weights for this run, mapped as clamp(72 + 10·z̄, 10, 100). It is an index, not a percentage. The primary path treats a missing active source as z = 0 (shrinkage to the board center). The sensitivity path drops missing sources and renormalizes the remaining weights; that peer is shown as a band only when coverage is incomplete. Sources that cover less than 80% of the scored set have their weight capped at 15% before renormalization so sparse boards cannot dominate. Stale or coverage-failed sources are dropped from the active set entirely. SWE-Atlas-QnA is not weighted (Hermes task mix is terminal/agent execution, not codebase QnA). BFCL-class and tau-bench were evaluated on 2026-08-30 against the freshness and frontier-hit floor and are absent from scoring.",
  "agent_arena_note": "Agent Arena is a rank-linear percentile, 100×(N−rank)/(N−1), so rank 1 is exactly 100% and the board ends near 0%. Against the scored set that compresses z to roughly ±1.35 by construction — much tighter than free metric scales. It carries weight but does not count toward the ≥4 direct-source floor.",
  "terminal_bench_note": "A dash means the model has no score on that active source. The primary Index uses z = 0 for that cell; the sensitivity band shows weight renormalization and appears only when coverage is incomplete. Imputed cells do not count toward the inclusion floor.",
  "sweatlas_qna_note": "A dash means the model has no score on that active source. The primary Index uses z = 0 for that cell; the sensitivity band shows weight renormalization and appears only when coverage is incomplete. Imputed cells do not count toward the inclusion floor.",
  "missing_note": "A dash means the model has no score on that active source. The primary Index uses z = 0 for that cell; the sensitivity band shows weight renormalization and appears only when coverage is incomplete. Imputed cells do not count toward the inclusion floor.",
  "source_eval_note": "Tool-use board audit 2026-08-30: BFCL v4 (BenchLM) was ≤7d fresh with 13 rows but 0 models in the Hermes frontier seed after name cleaning — fails ≥5 frontier hits. tau-bench (BenchLM) published an empty leaderboard (display-only / outdated tasks). Neither is a primary source. SWE-Atlas-QnA weight is 0 (not Hermes task mix). Terminal-Bench is pinned to release 4.0 (BenchLM mirror of tbench.ai; best harness/effort per model). Scores are not comparable to the prior TB 2.x column; cross-version fallback into one z-pool is forbidden.",
  "gpqa_note": "Only GPQA currently shows a ±95% Wald interval (n≈198 multiple-choice items). LMSYS is Elo, not a binomial accuracy; Agent Arena is a rank transform; BrowseComp, Terminal-Bench 4.0, DeepSWE, and HLE are publisher point estimates without a stable public n for the same binomial model. Extending CIs needs per-source n or SEs.",
  "ci_policy_note": "Only GPQA currently shows a ±95% Wald interval (n≈198 multiple-choice items). LMSYS is Elo, not a binomial accuracy; Agent Arena is a rank transform; BrowseComp, Terminal-Bench 4.0, DeepSWE, and HLE are publisher point estimates without a stable public n for the same binomial model. Extending CIs needs per-source n or SEs.",
  "picks_blurb": "Index ignores price. Default = best Index within 2× cheapest task $ · Cheap = lowest task $ · Frontier = best Index. Rules under Method.",
  "picks_formula": "Picks use the same top-15 board as the public leaderboard (models with ≥4 active directs, ranked by Hermes Index). The Hermes Index itself is capability-only and never uses price. Price enters only Cheap and Default. List $/1M means input+output USD per 1M tokens. Task $ means DeepSWE average $/task (real multi-step agent spend on that board). Frontier is the best Hermes Index (#1); the runner-up is the closest peer among ranks 2–5 within 2.0 Index points (higher Index then board rank on ties; no price), otherwise #2 by Index rank. Cheap is the lowest DeepSWE $/task inside that top-15 (list $/1M is only a tie-break or fallback when task $ is missing); its runner-up is the highest Index within 2× that task $ (else the next-cheapest by the same key). Default is the highest Hermes Index among models whose task $ is within 2× Cheap's task $ (list $ defines the band only when Cheap lacks task $), excluding Frontier #1 — the best daily driver still near the cheap frontier, not max Index÷$. The Default runner-up is the next-highest Index in that same band (if the band is a singleton, the next-cheapest model). Default may equal Cheap when Cheap also leads the band on Index; the rule never elevates a dominated model just to keep three category names distinct. Aggregate ≠ task proof.",
  "list_price_def": "input+output USD per 1M tokens",
  "index_label": "Hermes Index",
  "top_n": 15,
  "model_count_scored": 34,
  "model_count_shown": 15,
  "active_primary_benchmarks": [
    "lmsys",
    "hle",
    "gpqa_diamond",
    "terminalbench",
    "browsecomp",
    "deepswe"
  ],
  "models": [
    {
      "rank": 1,
      "model": "GPT-6",
      "score": 84.2,
      "score_impute": 84.2,
      "score_renorm": 84.2,
      "score_band_lo": 84.2,
      "score_band_hi": 84.2,
      "show_band": false,
      "coverage_star": false,
      "source_count": 6,
      "active_sources": 6,
      "coverage_label": "6/6 direct + Agent Arena",
      "coverage_weighted": 7,
      "coverage_weighted_total": 7,
      "missing_directs": [],
      "lmsys": 1479.7712562096513,
      "agent_arena": 97.61904761904762,
      "hle": 54.6802594995366,
      "gpqa_diamond": 96.2626262626263,
      "gpqa_ci95": 2.64,
      "terminalbench": 58.18,
      "browsecomp": 91.5,
      "sweatlas_qna": null,
      "deepswe": 74.11504424778761,
      "deepswe_task_cost": 4.429117203539823,
      "cost_in": "$10",
      "cost_out": "$50",
      "cost_compact": "10/50",
      "list_price": 60.0,
      "value_ratio": 19.011,
      "z_by_source": {
        "terminalbench": 2.2306,
        "deepswe": 0.8469,
        "browsecomp": 0.9478,
        "agent_lmsys": 1.096,
        "hle": 1.4651,
        "lmsys": 0.2585,
        "gpqa_diamond": 1.6862
      },
      "contrib_by_source": {
        "terminalbench": 0.3718,
        "deepswe": 0.1412,
        "browsecomp": 0.158,
        "agent_lmsys": 0.1827,
        "hle": 0.2442,
        "lmsys": 0.0287,
        "gpqa_diamond": 0.0937
      }
    },
    {
      "rank": 2,
      "model": "Claude Opus 5",
      "score": 83.34,
      "score_impute": 83.34,
      "score_renorm": 83.34,
      "score_band_lo": 83.34,
      "score_band_hi": 83.34,
      "show_band": false,
      "coverage_star": false,
      "source_count": 6,
      "active_sources": 6,
      "coverage_label": "6/6 direct + Agent Arena",
      "coverage_weighted": 7,
      "coverage_weighted_total": 7,
      "missing_directs": [],
      "lmsys": 1490.1319811614499,
      "agent_arena": 95.23809523809524,
      "hle": 54.8656163113994,
      "gpqa_diamond": 93.7373737373737,
      "gpqa_ci95": 3.37,
      "terminalbench": 53.94,
      "browsecomp": 90.8,
      "sweatlas_qna": null,
      "deepswe": 73.64864864864865,
      "deepswe_task_cost": 11.837583271396396,
      "cost_in": "$5.00",
      "cost_out": "$25.00",
      "cost_compact": "5/25",
      "list_price": 30.0,
      "value_ratio": 7.04,
      "z_by_source": {
        "terminalbench": 1.9649,
        "deepswe": 0.8119,
        "browsecomp": 0.8664,
        "agent_lmsys": 1.0117,
        "hle": 1.4951,
        "lmsys": 0.7386,
        "gpqa_diamond": 0.4818
      },
      "contrib_by_source": {
        "terminalbench": 0.3275,
        "deepswe": 0.1353,
        "browsecomp": 0.1444,
        "agent_lmsys": 0.1686,
        "hle": 0.2492,
        "lmsys": 0.0821,
        "gpqa_diamond": 0.0268
      }
    },
    {
      "rank": 3,
      "model": "Claude Fable 5.1",
      "score": 82.81,
      "score_impute": 82.81,
      "score_renorm": 88.22,
      "score_band_lo": 82.81,
      "score_band_hi": 88.22,
      "show_band": true,
      "coverage_star": true,
      "source_count": 4,
      "active_sources": 6,
      "coverage_label": "4/6 direct + Agent Arena",
      "coverage_weighted": 5,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "browsecomp",
        "deepswe"
      ],
      "lmsys": 1498.4730733605907,
      "agent_arena": 100.0,
      "hle": 59.1288229842447,
      "gpqa_diamond": 93.7373737373737,
      "gpqa_ci95": 3.37,
      "terminalbench": 57.88,
      "browsecomp": null,
      "sweatlas_qna": null,
      "deepswe": null,
      "deepswe_task_cost": null,
      "cost_in": "$10.00",
      "cost_out": "$50.00",
      "cost_compact": "10/50",
      "list_price": 60.0,
      "value_ratio": 1.38,
      "z_by_source": {
        "terminalbench": 2.2118,
        "deepswe": 0.0,
        "browsecomp": 0.0,
        "agent_lmsys": 1.1804,
        "hle": 2.1846,
        "lmsys": 1.1251,
        "gpqa_diamond": 0.4818
      },
      "contrib_by_source": {
        "terminalbench": 0.3686,
        "deepswe": 0.0,
        "browsecomp": 0.0,
        "agent_lmsys": 0.1967,
        "hle": 0.3641,
        "lmsys": 0.125,
        "gpqa_diamond": 0.0268
      }
    },
    {
      "rank": 4,
      "model": "Claude Fable 5",
      "score": 80.83,
      "score_impute": 80.83,
      "score_renorm": 82.6,
      "score_band_lo": 80.83,
      "score_band_hi": 82.6,
      "show_band": true,
      "coverage_star": false,
      "source_count": 5,
      "active_sources": 6,
      "coverage_label": "5/6 direct + Agent Arena",
      "coverage_weighted": 6,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "browsecomp"
      ],
      "lmsys": 1505.6827180827381,
      "agent_arena": 90.47619047619048,
      "hle": 55.468025949953706,
      "gpqa_diamond": 92.6262626262626,
      "gpqa_ci95": 3.64,
      "terminalbench": 44.55,
      "browsecomp": null,
      "sweatlas_qna": null,
      "deepswe": 69.91150442477876,
      "deepswe_task_cost": 13.414521495535714,
      "cost_in": "$10.00",
      "cost_out": "$50.00",
      "cost_compact": "10/50",
      "list_price": 60.0,
      "value_ratio": 6.026,
      "z_by_source": {
        "terminalbench": 1.3766,
        "deepswe": 0.5314,
        "browsecomp": 0.0,
        "agent_lmsys": 0.8431,
        "hle": 1.5925,
        "lmsys": 1.4592,
        "gpqa_diamond": -0.0482
      },
      "contrib_by_source": {
        "terminalbench": 0.2294,
        "deepswe": 0.0886,
        "browsecomp": 0.0,
        "agent_lmsys": 0.1405,
        "hle": 0.2654,
        "lmsys": 0.1621,
        "gpqa_diamond": -0.0027
      }
    },
    {
      "rank": 5,
      "model": "GPT-5.6 Sol",
      "score": 79.79,
      "score_impute": 79.79,
      "score_renorm": 79.79,
      "score_band_lo": 79.79,
      "score_band_hi": 79.79,
      "show_band": false,
      "coverage_star": false,
      "source_count": 6,
      "active_sources": 6,
      "coverage_label": "6/6 direct + Agent Arena",
      "coverage_weighted": 7,
      "coverage_weighted_total": 7,
      "missing_directs": [],
      "lmsys": 1483.474347288669,
      "agent_arena": 85.71428571428571,
      "hle": 49.4902687673772,
      "gpqa_diamond": 95.2525252525253,
      "gpqa_ci95": 2.96,
      "terminalbench": 37.27,
      "browsecomp": 92.2,
      "sweatlas_qna": null,
      "deepswe": 72.66666666666667,
      "deepswe_task_cost": 8.386436346666667,
      "cost_in": "$5.00",
      "cost_out": "$30.00",
      "cost_compact": "5/30",
      "list_price": 35.0,
      "value_ratio": 9.514,
      "z_by_source": {
        "terminalbench": 0.9204,
        "deepswe": 0.7382,
        "browsecomp": 1.0292,
        "agent_lmsys": 0.6745,
        "hle": 0.6258,
        "lmsys": 0.4301,
        "gpqa_diamond": 1.2044
      },
      "contrib_by_source": {
        "terminalbench": 0.1534,
        "deepswe": 0.123,
        "browsecomp": 0.1715,
        "agent_lmsys": 0.1124,
        "hle": 0.1043,
        "lmsys": 0.0478,
        "gpqa_diamond": 0.0669
      }
    },
    {
      "rank": 6,
      "model": "GLM-5.3",
      "score": 77.24,
      "score_impute": 77.24,
      "score_renorm": 78.29,
      "score_band_lo": 77.24,
      "score_band_hi": 78.29,
      "show_band": true,
      "coverage_star": false,
      "source_count": 5,
      "active_sources": 6,
      "coverage_label": "5/6 direct + Agent Arena",
      "coverage_weighted": 6,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "browsecomp"
      ],
      "lmsys": 1479.2111945529525,
      "agent_arena": 59.523809523809526,
      "hle": 55.468025949953706,
      "gpqa_diamond": 92.6262626262626,
      "gpqa_ci95": 3.64,
      "terminalbench": 41.82,
      "browsecomp": null,
      "sweatlas_qna": null,
      "deepswe": 68.95787139689578,
      "deepswe_task_cost": 3.9933584893126386,
      "cost_in": "$1.4",
      "cost_out": "$4.4",
      "cost_compact": "1.4/4.4",
      "list_price": 5.800000000000001,
      "value_ratio": 19.342,
      "z_by_source": {
        "terminalbench": 1.2055,
        "deepswe": 0.4598,
        "browsecomp": 0.0,
        "agent_lmsys": -0.2529,
        "hle": 1.5925,
        "lmsys": 0.2325,
        "gpqa_diamond": -0.0482
      },
      "contrib_by_source": {
        "terminalbench": 0.2009,
        "deepswe": 0.0766,
        "browsecomp": 0.0,
        "agent_lmsys": -0.0422,
        "hle": 0.2654,
        "lmsys": 0.0258,
        "gpqa_diamond": -0.0027
      }
    },
    {
      "rank": 7,
      "model": "Kimi K3",
      "score": 76.18,
      "score_impute": 76.18,
      "score_renorm": 77.01,
      "score_band_lo": 76.18,
      "score_band_hi": 77.01,
      "show_band": true,
      "coverage_star": false,
      "source_count": 5,
      "active_sources": 6,
      "coverage_label": "5/6 direct + Agent Arena",
      "coverage_weighted": 6,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "terminalbench"
      ],
      "lmsys": 1484.7659884633672,
      "agent_arena": 80.95238095238095,
      "hle": 46.8952734012975,
      "gpqa_diamond": 93.53535353535351,
      "gpqa_ci95": 3.43,
      "terminalbench": null,
      "browsecomp": 91.2,
      "sweatlas_qna": null,
      "deepswe": 68.51441241685144,
      "deepswe_task_cost": 4.654682129933482,
      "cost_in": "$3.00",
      "cost_out": "$15.00",
      "cost_compact": "3/15",
      "list_price": 18.0,
      "value_ratio": 16.366,
      "z_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.4265,
        "browsecomp": 0.9129,
        "agent_lmsys": 0.5059,
        "hle": 0.2061,
        "lmsys": 0.4899,
        "gpqa_diamond": 0.3854
      },
      "contrib_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.0711,
        "browsecomp": 0.1521,
        "agent_lmsys": 0.0843,
        "hle": 0.0343,
        "lmsys": 0.0544,
        "gpqa_diamond": 0.0214
      }
    },
    {
      "rank": 8,
      "model": "GPT-5.6 Terra",
      "score": 75.72,
      "score_impute": 75.72,
      "score_renorm": 75.72,
      "score_band_lo": 75.72,
      "score_band_hi": 75.72,
      "show_band": false,
      "coverage_star": false,
      "source_count": 6,
      "active_sources": 6,
      "coverage_label": "6/6 direct + Agent Arena",
      "coverage_weighted": 7,
      "coverage_weighted_total": 7,
      "missing_directs": [],
      "lmsys": 1466.0372736853974,
      "agent_arena": 47.61904761904762,
      "hle": 58.7117701575533,
      "gpqa_diamond": 93.4343434343434,
      "gpqa_ci95": 3.45,
      "terminalbench": 21.52,
      "browsecomp": 87.5,
      "sweatlas_qna": null,
      "deepswe": 69.62305986696231,
      "deepswe_task_cost": 4.945847263858093,
      "cost_in": "$2.00",
      "cost_out": "$12.00",
      "cost_compact": "2/12",
      "list_price": 14.0,
      "value_ratio": 15.31,
      "z_by_source": {
        "terminalbench": -0.0664,
        "deepswe": 0.5098,
        "browsecomp": 0.4826,
        "agent_lmsys": -0.6745,
        "hle": 2.1172,
        "lmsys": -0.3779,
        "gpqa_diamond": 0.3372
      },
      "contrib_by_source": {
        "terminalbench": -0.0111,
        "deepswe": 0.085,
        "browsecomp": 0.0804,
        "agent_lmsys": -0.1124,
        "hle": 0.3529,
        "lmsys": -0.042,
        "gpqa_diamond": 0.0187
      }
    },
    {
      "rank": 9,
      "model": "Gemini 3.8 Flash",
      "score": 75.38,
      "score_impute": 75.38,
      "score_renorm": 76.06,
      "score_band_lo": 75.38,
      "score_band_hi": 76.06,
      "show_band": true,
      "coverage_star": false,
      "source_count": 5,
      "active_sources": 6,
      "coverage_label": "5/6 direct + Agent Arena",
      "coverage_weighted": 6,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "browsecomp"
      ],
      "lmsys": 1493.0082765047282,
      "agent_arena": 69.04761904761905,
      "hle": 47.8220574606117,
      "gpqa_diamond": 95.2525252525253,
      "gpqa_ci95": 2.96,
      "terminalbench": 19.09,
      "browsecomp": null,
      "sweatlas_qna": null,
      "deepswe": 73.8255033557047,
      "deepswe_task_cost": 2.362349413758389,
      "cost_in": "$0.75",
      "cost_out": "$3.75",
      "cost_compact": "0.75/3.75",
      "list_price": 4.5,
      "value_ratio": 31.909,
      "z_by_source": {
        "terminalbench": -0.2187,
        "deepswe": 0.8252,
        "browsecomp": 0.0,
        "agent_lmsys": 0.0843,
        "hle": 0.356,
        "lmsys": 0.8719,
        "gpqa_diamond": 1.2044
      },
      "contrib_by_source": {
        "terminalbench": -0.0364,
        "deepswe": 0.1375,
        "browsecomp": 0.0,
        "agent_lmsys": 0.0141,
        "hle": 0.0593,
        "lmsys": 0.0969,
        "gpqa_diamond": 0.0669
      }
    },
    {
      "rank": 10,
      "model": "GPT-5.5",
      "score": 73.93,
      "score_impute": 73.93,
      "score_renorm": 74.32,
      "score_band_lo": 73.93,
      "score_band_hi": 74.32,
      "show_band": true,
      "coverage_star": false,
      "source_count": 5,
      "active_sources": 6,
      "coverage_label": "5/6 direct + Agent Arena",
      "coverage_weighted": 6,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "terminalbench"
      ],
      "lmsys": 1478.9203500467565,
      "agent_arena": 78.57142857142857,
      "hle": 45.783132530120504,
      "gpqa_diamond": 93.53535353535351,
      "gpqa_ci95": 3.43,
      "terminalbench": null,
      "browsecomp": 84.4,
      "sweatlas_qna": null,
      "deepswe": 67.03539823008849,
      "deepswe_task_cost": 7.226236674778761,
      "cost_in": "$5.00",
      "cost_out": "$30.00",
      "cost_compact": "5/30",
      "list_price": 35.0,
      "value_ratio": 10.231,
      "z_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.3155,
        "browsecomp": 0.1221,
        "agent_lmsys": 0.4216,
        "hle": 0.0262,
        "lmsys": 0.2191,
        "gpqa_diamond": 0.3854
      },
      "contrib_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.0526,
        "browsecomp": 0.0204,
        "agent_lmsys": 0.0703,
        "hle": 0.0044,
        "lmsys": 0.0243,
        "gpqa_diamond": 0.0214
      }
    },
    {
      "rank": 11,
      "model": "Claude Opus 4.8",
      "score": 73.87,
      "score_impute": 73.87,
      "score_renorm": 73.87,
      "score_band_lo": 73.87,
      "score_band_hi": 73.87,
      "show_band": false,
      "coverage_star": false,
      "source_count": 6,
      "active_sources": 6,
      "coverage_label": "6/6 direct + Agent Arena",
      "coverage_weighted": 7,
      "coverage_weighted_total": 7,
      "missing_directs": [],
      "lmsys": 1477.3004943535811,
      "agent_arena": 88.0952380952381,
      "hle": 48.6561631139944,
      "gpqa_diamond": 92.020202020202,
      "gpqa_ci95": 3.77,
      "terminalbench": 23.64,
      "browsecomp": 84.3,
      "sweatlas_qna": null,
      "deepswe": 58.97435897435898,
      "deepswe_task_cost": 13.222583593240094,
      "cost_in": "$5.00",
      "cost_out": "$25.00",
      "cost_compact": "5/25",
      "list_price": 30.0,
      "value_ratio": 5.587,
      "z_by_source": {
        "terminalbench": 0.0664,
        "deepswe": -0.2896,
        "browsecomp": 0.1105,
        "agent_lmsys": 0.7588,
        "hle": 0.4909,
        "lmsys": 0.144,
        "gpqa_diamond": -0.3372
      },
      "contrib_by_source": {
        "terminalbench": 0.0111,
        "deepswe": -0.0483,
        "browsecomp": 0.0184,
        "agent_lmsys": 0.1265,
        "hle": 0.0818,
        "lmsys": 0.016,
        "gpqa_diamond": -0.0187
      }
    },
    {
      "rank": 12,
      "model": "Muse Spark",
      "score": 73.48,
      "score_impute": 73.48,
      "score_renorm": 74.21,
      "score_band_lo": 73.48,
      "score_band_hi": 74.21,
      "show_band": true,
      "coverage_star": true,
      "source_count": 4,
      "active_sources": 6,
      "coverage_label": "4/6 direct + Agent Arena",
      "coverage_weighted": 5,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "terminalbench",
        "browsecomp"
      ],
      "lmsys": 1493.3279136166302,
      "agent_arena": 71.42857142857143,
      "hle": 48.7025023169602,
      "gpqa_diamond": 94.1414141414141,
      "gpqa_ci95": 3.27,
      "terminalbench": null,
      "browsecomp": null,
      "sweatlas_qna": null,
      "deepswe": 54.86725663716814,
      "deepswe_task_cost": 3.6955048180309733,
      "cost_in": "$1.25",
      "cost_out": "$4.25",
      "cost_compact": "1.25/4.25",
      "list_price": 5.5,
      "value_ratio": 19.884,
      "z_by_source": {
        "terminalbench": 0.0,
        "deepswe": -0.5978,
        "browsecomp": 0.0,
        "agent_lmsys": 0.1686,
        "hle": 0.4984,
        "lmsys": 0.8867,
        "gpqa_diamond": 0.6745
      },
      "contrib_by_source": {
        "terminalbench": 0.0,
        "deepswe": -0.0996,
        "browsecomp": 0.0,
        "agent_lmsys": 0.0281,
        "hle": 0.0831,
        "lmsys": 0.0985,
        "gpqa_diamond": 0.0375
      }
    },
    {
      "rank": 13,
      "model": "Claude Sonnet 5",
      "score": 71.6,
      "score_impute": 71.6,
      "score_renorm": 71.6,
      "score_band_lo": 71.6,
      "score_band_hi": 71.6,
      "show_band": false,
      "coverage_star": false,
      "source_count": 6,
      "active_sources": 6,
      "coverage_label": "6/6 direct + Agent Arena",
      "coverage_weighted": 7,
      "coverage_weighted_total": 7,
      "missing_directs": [],
      "lmsys": 1461.2069551775312,
      "agent_arena": 83.33333333333333,
      "hle": 48.7025023169602,
      "gpqa_diamond": 94.1414141414141,
      "gpqa_ci95": 3.27,
      "terminalbench": 12.42,
      "browsecomp": 84.7,
      "sweatlas_qna": null,
      "deepswe": 53.84615384615385,
      "deepswe_task_cost": 26.399858950791852,
      "cost_in": "$2.00",
      "cost_out": "$10.00",
      "cost_compact": "2/10",
      "list_price": 12.0,
      "value_ratio": 2.712,
      "z_by_source": {
        "terminalbench": -0.6366,
        "deepswe": -0.6745,
        "browsecomp": 0.157,
        "agent_lmsys": 0.5902,
        "hle": 0.4984,
        "lmsys": -0.6017,
        "gpqa_diamond": 0.6745
      },
      "contrib_by_source": {
        "terminalbench": -0.1061,
        "deepswe": -0.1124,
        "browsecomp": 0.0262,
        "agent_lmsys": 0.0984,
        "hle": 0.0831,
        "lmsys": -0.0669,
        "gpqa_diamond": 0.0375
      }
    },
    {
      "rank": 14,
      "model": "Claude Opus 4.7",
      "score": 71.2,
      "score_impute": 71.2,
      "score_renorm": 70.41,
      "score_band_lo": 70.41,
      "score_band_hi": 71.2,
      "show_band": true,
      "coverage_star": true,
      "source_count": 4,
      "active_sources": 6,
      "coverage_label": "4/6 direct",
      "coverage_weighted": 4,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "terminalbench",
        "deepswe"
      ],
      "lmsys": 1498.0981345958426,
      "agent_arena": null,
      "hle": 42.3076923076923,
      "gpqa_diamond": 91.4141414141414,
      "gpqa_ci95": 3.9,
      "terminalbench": null,
      "browsecomp": 79.3,
      "sweatlas_qna": null,
      "deepswe": null,
      "deepswe_task_cost": null,
      "cost_in": "$5.00",
      "cost_out": "$25.00",
      "cost_compact": "5/25",
      "list_price": 30.0,
      "value_ratio": 2.373,
      "z_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.0,
        "browsecomp": -0.471,
        "agent_lmsys": 0.0,
        "hle": -0.5358,
        "lmsys": 1.1077,
        "gpqa_diamond": -0.6263
      },
      "contrib_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.0,
        "browsecomp": -0.0785,
        "agent_lmsys": 0.0,
        "hle": -0.0893,
        "lmsys": 0.1231,
        "gpqa_diamond": -0.0348
      }
    },
    {
      "rank": 15,
      "model": "Claude Opus 4.6",
      "score": 71.09,
      "score_impute": 71.09,
      "score_renorm": 70.17,
      "score_band_lo": 70.17,
      "score_band_hi": 71.09,
      "show_band": true,
      "coverage_star": true,
      "source_count": 4,
      "active_sources": 6,
      "coverage_label": "4/6 direct",
      "coverage_weighted": 4,
      "coverage_weighted_total": 7,
      "missing_directs": [
        "terminalbench",
        "deepswe"
      ],
      "lmsys": 1500.9540717093487,
      "agent_arena": null,
      "hle": 39.9443929564411,
      "gpqa_diamond": 89.5959595959596,
      "gpqa_ci95": 4.25,
      "terminalbench": null,
      "browsecomp": 83.7,
      "sweatlas_qna": null,
      "deepswe": null,
      "deepswe_task_cost": null,
      "cost_in": "$5.00",
      "cost_out": "$25.00",
      "cost_compact": "5/25",
      "list_price": 30.0,
      "value_ratio": 2.37,
      "z_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.0,
        "browsecomp": 0.0407,
        "agent_lmsys": 0.0,
        "hle": -0.9181,
        "lmsys": 1.2401,
        "gpqa_diamond": -1.4935
      },
      "contrib_by_source": {
        "terminalbench": 0.0,
        "deepswe": 0.0,
        "browsecomp": 0.0068,
        "agent_lmsys": 0.0,
        "hle": -0.153,
        "lmsys": 0.1378,
        "gpqa_diamond": -0.083
      }
    }
  ],
  "recommendations": [
    {
      "key": "default",
      "title": "Default",
      "model": "GLM-5.3",
      "rank": 6,
      "score": 77.24,
      "cost_compact": "1.4/4.4",
      "deepswe_task_cost": 3.9933584893126386,
      "value_ratio": 19.342,
      "show_band": true,
      "coverage_star": false,
      "alt": {
        "model": "Kimi K3",
        "rank": 7,
        "score": 76.18,
        "cost_compact": "3/15",
        "deepswe_task_cost": 4.654682129933482,
        "value_ratio": 16.366,
        "show_band": true,
        "coverage_star": false
      }
    },
    {
      "key": "cheap",
      "title": "Cheap",
      "model": "Gemini 3.8 Flash",
      "rank": 9,
      "score": 75.38,
      "cost_compact": "0.75/3.75",
      "deepswe_task_cost": 2.362349413758389,
      "value_ratio": 31.909,
      "show_band": true,
      "coverage_star": false,
      "alt": {
        "model": "GPT-6",
        "rank": 1,
        "score": 84.2,
        "cost_compact": "10/50",
        "deepswe_task_cost": 4.429117203539823,
        "value_ratio": 19.011,
        "show_band": false,
        "coverage_star": false
      }
    },
    {
      "key": "frontier",
      "title": "Frontier",
      "model": "GPT-6",
      "rank": 1,
      "score": 84.2,
      "cost_compact": "10/50",
      "deepswe_task_cost": 4.429117203539823,
      "value_ratio": 19.011,
      "show_band": false,
      "coverage_star": false,
      "alt": {
        "model": "Claude Opus 5",
        "rank": 2,
        "score": 83.34,
        "cost_compact": "5/25",
        "deepswe_task_cost": 11.837583271396396,
        "value_ratio": 7.04,
        "show_band": false,
        "coverage_star": false
      }
    }
  ],
  "sources": [
    {
      "key": "lmsys",
      "label": "LMSYS Text Arena",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 15,
      "url": "https://arena.ai/leaderboard/text/overall"
    },
    {
      "key": "hle",
      "label": "HLE",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 16,
      "url": "https://artificialanalysis.ai/leaderboards/models"
    },
    {
      "key": "gpqa_diamond",
      "label": "GPQA Diamond",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 16,
      "url": "https://artificialanalysis.ai/leaderboards/models"
    },
    {
      "key": "terminalbench",
      "label": "Terminal-Bench 4.0",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 10,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-4"
    },
    {
      "key": "browsecomp",
      "label": "BrowseComp",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 7,
      "url": "https://benchlm.ai/benchmarks/browsecomp"
    },
    {
      "key": "deepswe",
      "label": "DeepSWE",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 14,
      "url": "https://deepswe.datacurve.ai/"
    },
    {
      "key": "agent_lmsys",
      "label": "Agent Arena",
      "as_of": "2026-09-25",
      "status": "active",
      "frontier_hits": 17,
      "url": "https://arena.ai/leaderboard/agent"
    }
  ],
  "notes": [
    "Hermes-oriented composite for personal agents — not a neutral AGI ranking.",
    "≥4 active direct sources required (Agent Arena does not count toward the floor).",
    "List $/1M = input+output USD per 1M tokens. DeepSWE $ = avg $/task at best-effort config.",
    "* Fewest active benchmarks on this board.",
    "Aggregate ≠ task proof."
  ],
  "coverage_star_note": "* Fewest active benchmarks on this board."
}
