Skip to content

Commit be98f9d

Browse files
bai-uipathclaude
andauthored
fix(pricing): refresh the rate card and add gemini 3.7/3.8 Flash (#155)
* fix(pricing): refresh the rate card and add gemini 3.7/3.8 Flash Re-verified every row against the vendor cards on 2026-09-03. Four rows were stale and three models were missing. Repriced: - claude-sonnet-5 $3/$15 -> $2/$10. The $2/$10 introductory rate is now the standard price and the 2026-09-01 increase was cancelled, so the deliberate hedge against it lapsing was overstating the nightly's own model by 50%. - gpt-5.6-sol $5/$30 -> $4/$20. Sol took its own cut after the 2026-07-30 Terra/Luna one. - z-ai/glm-5.2 and deepseek/deepseek-v4-pro, read from OpenRouter's live /api/v1/models feed. Both had drifted up; deepseek's cache read was low by ~24x. Added: - gemini-3.8-flash and gemini-3.7-flash. 3.8 is now the ANTIGRAVITY_MODEL in both variable stores, and an unpriced model books no cost at all, so an antigravity run reported cost_complete=false with a null bill. - claude-fable-5-1, which prices cache hits at 0.025x input rather than the 0.1x every other Claude model uses. The three Bedrock open-weight rows are carried forward untouched: AWS does not publish eu-north-1 figures for them, and the comment there warns against substituting the US column. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * docs(pricing): trim the rate-card comments to what affects an edit Drop the price-change narratives. What each row USED to cost is git history; the comments now carry only the traps a future edit can fall into. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
1 parent 0faf34a commit be98f9d

4 files changed

Lines changed: 43 additions & 23 deletions

File tree

‎evalboard/lib/pricing.ts‎

Lines changed: 11 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -24,6 +24,9 @@ export const PRICING: Record<string, Pricing> = {
2424
// alias must never inherit the older tier. Undated aliases each need their
2525
// own key: resolvePricing's fallback only strips a trailing date (dated →
2626
// undated), it cannot invent one.
27+
// Fable 5.1 prices cache hits at 0.025x input, not the 0.1x every other
28+
// Claude model uses. Fable 5 pays $1 on the identical $10 base.
29+
"claude-fable-5-1": p(10, 50, 12.5, 0.25),
2730
"claude-fable-5": p(10, 50, 12.5, 1),
2831
"claude-opus-5": p(5, 25, 6.25, 0.5),
2932
"claude-opus-4-8": p(5, 25, 6.25, 0.5),
@@ -34,7 +37,8 @@ export const PRICING: Record<string, Pricing> = {
3437
"claude-opus-4-1": p(15, 75, 18.75, 1.5),
3538
"claude-opus-4": p(15, 75, 18.75, 1.5),
3639
"claude-opus-4-20250514": p(15, 75, 18.75, 1.5),
37-
"claude-sonnet-5": p(3, 15, 3.75, 0.3),
40+
// $2/$10, not the $3/$15 that 4.6 and earlier pay.
41+
"claude-sonnet-5": p(2, 10, 2.5, 0.2),
3842
"claude-sonnet-4-6": p(3, 15, 3.75, 0.3),
3943
"claude-sonnet-4-5": p(3, 15, 3.75, 0.3),
4044
"claude-sonnet-4-5-20250929": p(3, 15, 3.75, 0.3),
@@ -64,8 +68,8 @@ export const PRICING: Record<string, Pricing> = {
6468
"gpt-5.2-codex": p(1.75, 14, 1.75, 0.175),
6569
"gpt-5.4": p(2.5, 15, 2.5, 0.25),
6670
"gpt-5.5": p(5, 30, 5, 0.5),
67-
// Terra and Luna repriced 2026-07-30 (-20% / -80%); post-cut rates.
68-
"gpt-5.6-sol": p(5, 30, 5, 0.5),
71+
// Sol's rate is promotional through at least 2026-11-21.
72+
"gpt-5.6-sol": p(4, 20, 4, 0.4),
6973
"gpt-5.6-terra": p(2, 12, 2, 0.2),
7074
"gpt-5.6-luna": p(0.2, 1.2, 0.2, 0.02),
7175
// Google Gemini (AntigravityAgent). Gemini bills no separate cache-write
@@ -75,6 +79,10 @@ export const PRICING: Record<string, Pricing> = {
7579
"gemini-3-pro-preview": p(2, 12, 2, 0.2),
7680
"gemini-3.1-pro-preview": p(2, 12, 2, 0.2),
7781
"gemini-3.1-pro-preview-customtools": p(2, 12, 2, 0.2),
82+
// 3.6 / 3.7 / 3.8 Flash share one rate card. List rates; Google is
83+
// discounting all three by half through 2026-12-31.
84+
"gemini-3.8-flash": p(1.5, 7.5, 1.5, 0.15),
85+
"gemini-3.7-flash": p(1.5, 7.5, 1.5, 0.15),
7886
"gemini-3.6-flash": p(1.5, 7.5, 1.5, 0.15),
7987
"gemini-3.5-flash": p(1.5, 9, 1.5, 0.15),
8088
"gemini-3.5-flash-lite": p(0.3, 2.5, 0.3, 0.03),

‎src/coder_eval/pricing.py‎

Lines changed: 27 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -2,9 +2,11 @@
22
33
Anthropic/OpenAI/Google built-in rates; plugins contribute additional rates via
44
``register_pricing()``. Prices are per million tokens (MTok).
5-
Sources: https://claude.com/pricing#api, https://developers.openai.com/api/docs/pricing,
6-
https://ai.google.dev/gemini-api/docs/pricing (all verified 2026-07-29;
7-
the GPT-5.6 rows re-verified 2026-08-09 after the 2026-07-30 Terra/Luna cut).
5+
Sources: https://platform.claude.com/docs/en/about-claude/pricing,
6+
https://developers.openai.com/api/docs/pricing,
7+
https://ai.google.dev/gemini-api/docs/pricing, and OpenRouter's live
8+
``/api/v1/models`` (every row re-verified 2026-09-03, except the Bedrock
9+
open-weight block: AWS publishes no eu-north-1 figures for those three).
810
"""
911

1012
from collections.abc import Iterable
@@ -21,9 +23,12 @@ class ModelPricing:
2123
cache_read_per_mtok: float # prompt caching read
2224

2325

24-
# Official vendor rate cards, verified 2026-07-29 (GPT-5.6 rows: 2026-08-09).
26+
# Official vendor rate cards, verified 2026-09-03.
2527
# Key: CLI model name (before gateway mapping)
2628
_PRICING: dict[str, ModelPricing] = {
29+
# Fable 5.1 (and Mythos 5.1) price cache hits at 0.025x input, not the 0.1x
30+
# every other Claude model uses. Fable 5 pays $1 on the identical $10 base.
31+
"claude-fable-5-1": ModelPricing(10.0, 50.0, 12.50, 0.25),
2732
"claude-fable-5": ModelPricing(10.0, 50.0, 12.50, 1.0),
2833
# Opus 4.5 and later dropped to $5/$25; 4.1 and 4 keep the old $15/$75. The
2934
# version boundary is the price boundary: a newer Opus is not the dearer one.
@@ -36,10 +41,9 @@ class ModelPricing:
3641
"claude-opus-4-1": ModelPricing(15.0, 75.0, 18.75, 1.50),
3742
"claude-opus-4": ModelPricing(15.0, 75.0, 18.75, 1.50),
3843
"claude-opus-4-20250514": ModelPricing(15.0, 75.0, 18.75, 1.50),
39-
# Standard $3/$15, not the $2/$10 promo running through 2026-08-31: a static
40-
# table cannot express a window, and overstating for a few weeks beats
41-
# understating indefinitely after it lapses.
42-
"claude-sonnet-5": ModelPricing(3.0, 15.0, 3.75, 0.30),
44+
# $2/$10, NOT the $3/$15 that Sonnet 4.6 and earlier pay. Do not copy the
45+
# 4.x row onto it.
46+
"claude-sonnet-5": ModelPricing(2.0, 10.0, 2.50, 0.20),
4347
"claude-sonnet-4-6": ModelPricing(3.0, 15.0, 3.75, 0.30),
4448
"claude-sonnet-4-5": ModelPricing(3.0, 15.0, 3.75, 0.30),
4549
"claude-sonnet-4-5-20250929": ModelPricing(3.0, 15.0, 3.75, 0.30),
@@ -80,19 +84,21 @@ class ModelPricing:
8084
"gpt-5.4-mini": ModelPricing(0.75, 4.5, 0.75, 0.075),
8185
"gpt-5.4-nano": ModelPricing(0.20, 1.25, 0.20, 0.02),
8286
# GPT-5.6: sol flagship / terra balanced (Codex default) / luna economy.
83-
# Terra and Luna were REPRICED on 2026-07-30 (-20% and -80%); these are the
84-
# post-cut rates. The pre-cut $2.50/$15 and $1.00/$6 are what a historical run
85-
# was actually billed, but this table is a single current-rate card with no
86-
# notion of an effective date — so old runs re-price low, the same tradeoff
87-
# the Sonnet promo comment above already accepts.
88-
"gpt-5.6-sol": ModelPricing(5.0, 30.0, 5.0, 0.50),
87+
# This table is a single current-rate card with no notion of an effective
88+
# date, so a repriced model makes historical runs re-price at today's rate.
89+
# Sol's rate is promotional through at least 2026-11-21; re-check then.
90+
"gpt-5.6-sol": ModelPricing(4.0, 20.0, 4.0, 0.40),
8991
"gpt-5.6-terra": ModelPricing(2.0, 12.0, 2.0, 0.20),
9092
"gpt-5.6-luna": ModelPricing(0.20, 1.20, 0.20, 0.02),
9193
# Google Gemini (AntigravityAgent, via the Gemini Developer API), keyed on the
9294
# literal ids the ListModels endpoint returns. No cache-write fee, so
9395
# cache_write == input (unused: the agent maps cache_creation_tokens to 0).
9496
# CAVEAT: Pro's >200K-token tier costs more ($4/$18, $0.40 cached), so a
9597
# very-large-context run reads low.
98+
# 3.6 / 3.7 / 3.8 Flash share one rate card. These are list rates; Google is
99+
# discounting all three by half through 2026-12-31.
100+
"gemini-3.8-flash": ModelPricing(1.5, 7.5, 1.5, 0.15),
101+
"gemini-3.7-flash": ModelPricing(1.5, 7.5, 1.5, 0.15),
96102
"gemini-3.6-flash": ModelPricing(1.5, 7.5, 1.5, 0.15),
97103
"gemini-3.5-flash": ModelPricing(1.5, 9.0, 1.5, 0.15),
98104
"gemini-3.5-flash-lite": ModelPricing(0.30, 2.5, 0.30, 0.03),
@@ -113,10 +119,14 @@ class ModelPricing:
113119
"moonshotai.kimi-k2.5": ModelPricing(0.72, 3.6, 0.72, 0.0),
114120
# OpenRouter models. These providers cache prefixes implicitly (no
115121
# cache_control, no write fee), so cache-creation is priced at input (unused)
116-
# and cache-read at OpenRouter's published input_cache_read rate.
122+
# and cache-read at OpenRouter's published input_cache_read rate, read from
123+
# the live /api/v1/models catalogue. Headline rates only: OpenRouter routes
124+
# per request, so the real bill depends on the provider a call lands on —
125+
# which is why the litellm path captures actual per-call cost proxy-side and
126+
# overrides these (litellm_cost.apply_actual_cost). Static fallback.
117127
"moonshotai/kimi-k3": ModelPricing(3.0, 15.0, 3.0, 0.30),
118-
"z-ai/glm-5.2": ModelPricing(0.7168, 2.2528, 0.7168, 0.13312),
119-
"deepseek/deepseek-v4-pro": ModelPricing(0.435, 0.87, 0.435, 0.003625),
128+
"z-ai/glm-5.2": ModelPricing(0.966, 3.036, 0.966, 0.1932),
129+
"deepseek/deepseek-v4-pro": ModelPricing(1.030776, 2.061552, 1.030776, 0.085898),
120130
}
121131

122132

‎tests/test_antigravity_agent.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -213,6 +213,8 @@ def test_tool_name_map_covers_core_builtins():
213213
"gemini-3.1-pro-preview",
214214
"gemini-3.1-pro-preview-customtools",
215215
"gemini-3-pro-preview",
216+
"gemini-3.8-flash",
217+
"gemini-3.7-flash",
216218
"gemini-3.6-flash",
217219
"gemini-3.5-flash",
218220
"gemini-3.5-flash-lite",

‎tests/test_cost_accounting_paths.py‎

Lines changed: 3 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -259,7 +259,7 @@ def test_simulator_prices_at_the_route_model_not_the_subject(self):
259259
"""UserSimulator pins model=None, so it bills at BEDROCK_MODEL.
260260
261261
A task that pins ``agent.model`` would otherwise mis-bill every simulated
262-
row. Here the subject is sonnet-5 ($3/$15) while the route is haiku-4.5
262+
row. Here the subject is sonnet-5 ($2/$10) while the route is haiku-4.5
263263
($1/$5), so the simulator must cost the haiku rate.
264264
"""
265265
result = self._simulated(bedrock_model="claude-haiku-4-5-20251001")
@@ -269,8 +269,8 @@ def test_simulator_prices_at_the_route_model_not_the_subject(self):
269269
def test_simulator_falls_back_to_the_subject_model_off_bedrock(self):
270270
"""A non-Bedrock route names no model on the record; the subject's is the best available."""
271271
result = self._simulated()
272-
# sonnet-5 at $3/$15 per MTok.
273-
assert eval_result_to_task_dict(result)["simulator_cost_usd"] == pytest.approx(3.0 + 1.5)
272+
# sonnet-5 at $2/$10 per MTok: 1M uncached input + 100K output.
273+
assert eval_result_to_task_dict(result)["simulator_cost_usd"] == pytest.approx(2.0 + 1.0)
274274

275275
def test_single_shot_row_has_no_simulator_cost(self):
276276
result = _result([_turn(1, TokenUsage(uncached_input_tokens=10, output_tokens=1, total_cost_usd=0.1))])

0 commit comments

Comments
 (0)