Skip to content

Commit 7ccf787

Browse files
committed
Drop the models Fireworks decommissions on 2026-09-25, and alias their names
Fireworks is removing DeepSeek V4 Pro 0813, V4 Flash 0731, V4 Flash Vision Exp and GLM 5.2 from serverless on 2026-09-25. V4.1 Flash replaces all three DeepSeek builds, and Fireworks states it outperforms V4 Pro 0813 on their benchmarks; GLM 5.3 replaces 5.2. This reverses yesterday's decision to keep GLM 5.2 pinnable. That was the right call when 5.2 looked merely superseded; it is the wrong one now that it has a removal date, because it would hand out a pin that breaks in two weeks. Deleting the rows outright is not safe either. The gateway answers an unknown label with a hard 400, so anyone pinned to a deleted model would break the moment we shipped rather than on the 25th. So model entries gained an `aliases` list: retired labels and ids resolve to their replacement's entry. A pin naming a decommissioned model migrates itself, which is what the vendor is doing anyway. Aliases live in their own index and resolve only after the live id and label maps miss, so a dead name can never shadow a live model, and the catalog panics at load if one tries. Consequence worth stating plainly: a retired id now PRICES as its replacement, because it resolves to that entry. That is correct for anything served from here on, and it is why the retired-id rate rows added earlier today came back out rather than sitting in the file describing a price nothing charges. The compact-budget and cheap-lane tests were asserting hardcoded 1,000,000 windows for models whose window moved. Both now derive the window from the catalog, which is the property they were written to defend in the first place.
1 parent 1fc4b41 commit 7ccf787

5 files changed

Lines changed: 107 additions & 139 deletions

File tree

‎catalog/catalog.go‎

Lines changed: 28 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -59,6 +59,14 @@ type CatalogModel struct {
5959
// Empty means "no floor": send whatever the effort maps to.
6060
MinReasoningEffort string `json:"min_reasoning_effort,omitempty"`
6161

62+
// Aliases are RETIRED labels and ids that must keep resolving to this entry.
63+
// When a vendor decommissions a model, deleting its row would make every pin
64+
// naming it a hard "unknown model" 400 at the gateway; listing the dead name
65+
// here migrates those pins onto the replacement instead, which is exactly
66+
// what the vendor's own migration does. A real label always wins over an
67+
// alias, so an alias can never shadow a live model.
68+
Aliases []string `json:"aliases,omitempty"`
69+
6270
// Fallback is the model's mid-turn failure chain, in LABELS: who covers
6371
// when this model errors after transport retries, walked IN ORDER by the
6472
// CLI's recovery executor (availability/billing filtering happens at walk
@@ -130,6 +138,7 @@ type loadedCatalog struct {
130138
file catalogFile
131139
byID map[string]CatalogModel
132140
byLabel map[string]CatalogModel
141+
byAlias map[string]CatalogModel
133142
}
134143

135144
func mustLoadModelCatalog(data []byte) *loadedCatalog {
@@ -141,6 +150,7 @@ func mustLoadModelCatalog(data []byte) *loadedCatalog {
141150
file: f,
142151
byID: make(map[string]CatalogModel, len(f.Models)),
143152
byLabel: make(map[string]CatalogModel, len(f.Models)),
153+
byAlias: make(map[string]CatalogModel),
144154
}
145155
for _, m := range f.Models {
146156
c.byID[m.ID] = m
@@ -154,6 +164,20 @@ func mustLoadModelCatalog(data []byte) *loadedCatalog {
154164
}
155165
c.byLabel[m.Label] = m
156166
}
167+
// Aliases resolve in a SEPARATE pass and a separate map: a retired name must
168+
// never shadow a live label, and two dead models may legitimately migrate to
169+
// the same replacement.
170+
for _, m := range f.Models {
171+
for _, a := range m.Aliases {
172+
if _, live := c.byID[a]; live {
173+
panic(fmt.Sprintf("common: alias %q on %q is a live model id", a, m.ID))
174+
}
175+
if _, live := c.byLabel[a]; live {
176+
panic(fmt.Sprintf("common: alias %q on %q is a live model label", a, m.ID))
177+
}
178+
c.byAlias[a] = m
179+
}
180+
}
157181
return c
158182
}
159183

@@ -203,7 +227,10 @@ func LookupModel(idOrLabel string) (CatalogModel, bool) {
203227
if m, ok := modelCatalog.byID[idOrLabel]; ok {
204228
return m, true
205229
}
206-
m, ok := modelCatalog.byLabel[idOrLabel]
230+
if m, ok := modelCatalog.byLabel[idOrLabel]; ok {
231+
return m, true
232+
}
233+
m, ok := modelCatalog.byAlias[idOrLabel]
207234
return m, ok
208235
}
209236

‎catalog/models.json‎

Lines changed: 14 additions & 64 deletions
Original file line numberDiff line numberDiff line change
@@ -321,7 +321,7 @@
321321
"price_in": 1.4,
322322
"price_out": 4.4,
323323
"price_cache_read": 0.26,
324-
"_note": "GLM-5.3 (Z.ai, Fireworks 2026-08-28). Supersedes GLM-5.2 as the pinnable GLM and as the Fireworks intra-lane fallback target. 743B params, 1M context, 131,072 max output, function calling, no vision; gains over 5.2 are post-training only. Official Standard-tier rate verified on the Fireworks model page 2026-09-11: $1.40 in / $0.26 cached / $4.40 out. Same headline price as 5.2, but the cached rate is nearly double, so it is declared explicitly.",
324+
"_note": "GLM-5.3 (Z.ai, Fireworks 2026-08-28). Supersedes GLM-5.2 as the pinnable GLM and as the Fireworks intra-lane fallback target. 743B params, 1M context, 131,072 max output, function calling, no vision; gains over 5.2 are post-training only. Official Standard-tier rate verified on the Fireworks model page 2026-09-11: $1.40 in / $0.26 cached / $4.40 out. Same headline price as 5.2, but the cached rate is nearly double, so it is declared explicitly. Fireworks decommissions GLM 5.2 from serverless on 2026-09-25 and names 5.3 as its replacement, so 5.2's label and id are aliased here.",
325325
"pinnable": true,
326326
"group": "GLM",
327327
"desc": "Open-weight coding flagship",
@@ -335,6 +335,10 @@
335335
"fallback": [
336336
"terra",
337337
"luna"
338+
],
339+
"aliases": [
340+
"glm-5p2",
341+
"accounts/fireworks/models/glm-5p2"
338342
]
339343
},
340344
{
@@ -363,32 +367,6 @@
363367
"glm-5p3"
364368
]
365369
},
366-
{
367-
"id": "accounts/fireworks/models/glm-5p2",
368-
"label": "glm-5p2",
369-
"vendor": "fireworks",
370-
"window": 1000000,
371-
"vision": false,
372-
"reasoning": true,
373-
"_note": "GLM-5.2 (Z.ai, Fireworks 2026-06-17). SUPERSEDED by glm-5p3 2026-09-11 and demoted out of the pickers; still deployed on Fireworks, so the entry stays for anyone already pinned to it and to keep its metering exact. Official rate 2026-09-11: $1.40 / $0.14 cached / $4.40 \u2014 now declared here rather than inherited from the family floor. 1M context, function calling, no vision.",
374-
"pinnable": false,
375-
"group": "GLM",
376-
"desc": "Open-weight coding flagship",
377-
"name": "GLM-5.2",
378-
"www": {
379-
"chat": false,
380-
"order": 30,
381-
"description": "Open-weight coding flagship (Z.ai) with a 1M context window \u2014 the same cheap lane the memcode CLI runs on."
382-
},
383-
"max_output": 32000,
384-
"fallback": [
385-
"terra",
386-
"glm-5p3"
387-
],
388-
"price_in": 1.4,
389-
"price_out": 4.4,
390-
"price_cache_read": 0.14
391-
},
392370
{
393371
"id": "accounts/fireworks/models/qwen3p8-max",
394372
"label": "qwen3p8-max",
@@ -456,40 +434,14 @@
456434
"glm-5p3"
457435
]
458436
},
459-
{
460-
"id": "accounts/fireworks/models/deepseek-v4-pro-0813",
461-
"label": "deepseek-v4-pro",
462-
"vendor": "fireworks",
463-
"window": 1048576,
464-
"vision": false,
465-
"reasoning": true,
466-
"_note": "DeepSeek V4 Pro 0813 (Fireworks official release, supersedes the earlier deepseek-v4-pro preview). Updated 2026-08-26 after Fireworks model page showed serverless Ready with path accounts/fireworks/models/deepseek-v4-pro-0813, 1,048,576 context (the value Fireworks /models reports; the entry previously said 1,040,000), function calling, $1.32/$3.96 per 1M, cached input $0.044. Label stays deepseek-v4-pro so existing pins keep following the latest Pro release.",
467-
"pinnable": true,
468-
"group": "DeepSeek",
469-
"desc": "Open-weight reasoning",
470-
"name": "DeepSeek V4 Pro",
471-
"www": {
472-
"chat": true,
473-
"order": 25,
474-
"description": "DeepSeek's official V4 Pro reasoning and coding model with a 1.04M context window."
475-
},
476-
"max_output": 32000,
477-
"fallback": [
478-
"terra",
479-
"glm-5p3"
480-
],
481-
"price_in": 1.32,
482-
"price_out": 3.96,
483-
"price_cache_read": 0.044
484-
},
485437
{
486438
"id": "accounts/fireworks/models/deepseek-v4p1-flash",
487439
"label": "deepseek-v4-flash",
488440
"vendor": "fireworks",
489441
"window": 1048576,
490442
"vision": true,
491443
"reasoning": true,
492-
"_note": "DeepSeek V4.1 Flash (Fireworks serverless, 2026-09-10). Supersedes deepseek-v4-flash-0731. 1M context, function calling, and vision, which 0731 did not have. Price is unchanged from 0731 and verified on the Fireworks model page 2026-09-11: $0.22 in, $0.007 cached, $0.66 out.",
444+
"_note": "DeepSeek V4.1 Flash (Fireworks serverless, 2026-09-10). 1M context, function calling, vision. Official rate 2026-09-11: $0.22 in, $0.007 cached, $0.66 out. Fireworks decommissions V4 Flash 0731, V4 Pro 0813 and V4 Flash Vision Exp from serverless on 2026-09-25 and names this model as the replacement for all three, so their labels and ids are aliased here and existing pins migrate instead of 400ing. Fireworks states V4.1 Flash outperforms V4 Pro 0813 on official benchmarks and carries the same vision capability as V4 Flash Vision Exp.",
493445
"pinnable": true,
494446
"group": "DeepSeek",
495447
"desc": "Fast reasoning",
@@ -506,7 +458,13 @@
506458
],
507459
"price_in": 0.22,
508460
"price_out": 0.66,
509-
"price_cache_read": 0.007
461+
"price_cache_read": 0.007,
462+
"aliases": [
463+
"deepseek-v4-pro",
464+
"accounts/fireworks/models/deepseek-v4-pro-0813",
465+
"accounts/fireworks/models/deepseek-v4-flash-0731",
466+
"accounts/fireworks/models/deepseek-v4-flash-vision-exp"
467+
]
510468
},
511469
{
512470
"id": "accounts/fireworks/models/minimax-m3",
@@ -685,7 +643,7 @@
685643
"price_in": 0.22,
686644
"price_out": 0.66,
687645
"price_cache_read": 0.007,
688-
"_note": "Legacy deepseek-v4-flash-0731, superseded by deepseek-v4p1-flash but still deployed and still pinnable. Without this row it falls to the Fireworks floor and bills 6x its real rate. Does not match deepseek-v4p1-flash."
646+
"_note": "Catches a drifted deepseek-v4-flash-* id that is not explicitly aliased. The 0731 build itself resolves through the alias on deepseek-v4p1-flash and prices there."
689647
},
690648
{
691649
"match": [
@@ -713,14 +671,6 @@
713671
"price_out": 4.4,
714672
"price_cache_read": 0.26
715673
},
716-
{
717-
"match": [
718-
"glm-5p2"
719-
],
720-
"price_in": 1.4,
721-
"price_out": 4.4,
722-
"price_cache_read": 0.14
723-
},
724674
{
725675
"match": [
726676
"glm",

‎catalog/pricing_test.go‎

Lines changed: 46 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -26,7 +26,9 @@ func TestModelPricingRealIDs(t *testing.T) {
2626
{"accounts/fireworks/models/kimi-k3", 3.00, 15.00}, // K3's own Fireworks headline card ($3/$15), NOT the kimi family rule
2727
{"accounts/fireworks/models/qwen3p8-max", 2.00, 6.00},
2828
{"accounts/fireworks/models/qwen3p8-2p4t-a95b", 2.00, 6.00}, // legacy path
29-
{"accounts/fireworks/models/deepseek-v4-pro-0813", 1.32, 3.96},
29+
// Decommissioned 2026-09-25 and aliased onto V4.1 Flash, so they price as Flash.
30+
{"accounts/fireworks/models/deepseek-v4-pro-0813", 0.22, 0.66},
31+
{"deepseek-v4-pro", 0.22, 0.66},
3032
{"accounts/fireworks/models/deepseek-v4p1-flash", 0.22, 0.66},
3133
// Superseded but still deployed on Fireworks: without its own rate row this
3234
// falls to the $1.40/$4.40 Fireworks floor and bills 6x.
@@ -62,7 +64,8 @@ func TestModelPricingByLabel(t *testing.T) {
6264
}{
6365
{"sol", 5}, {"terra", 2}, {"luna", 0.2},
6466
{"glm-5p2", 1.40}, {"kimi-k3", 3.00}, {"qwen3p8-max", 2.00},
65-
{"deepseek-v4-pro", 1.32}, {"deepseek-v4-flash", 0.22},
67+
// deepseek-v4-pro is an alias onto V4.1 Flash as of the 2026-09-25 decommission.
68+
{"deepseek-v4-pro", 0.22}, {"deepseek-v4-flash", 0.22},
6669
{"gemini-flash-lite", 0.3},
6770
}
6871
for _, c := range cases {
@@ -71,8 +74,9 @@ func TestModelPricingByLabel(t *testing.T) {
7174
}
7275
}
7376
// Windows resolve by label too (the footer meter sees labels, not raw ids).
74-
if got := ContextWindow("glm-5p2"); got != 1_000_000 {
75-
t.Errorf("ContextWindow(label glm-5p2) = %d, want 1M", got)
77+
// glm-5p2 is an alias onto GLM 5.3, whose window is 1,048,576.
78+
if got := ContextWindow("glm-5p2"); got != 1_048_576 {
79+
t.Errorf("ContextWindow(label glm-5p2) = %d, want 1048576", got)
7680
}
7781
if got := ContextWindow("qwen3p8-max"); got != 262_144 {
7882
t.Errorf("ContextWindow(label qwen3p8-max) = %d, want 262144", got)
@@ -84,15 +88,15 @@ func TestContextWindowFireworks(t *testing.T) {
8488
id string
8589
want int
8690
}{
87-
{"accounts/fireworks/models/glm-5p2", 1_000_000}, // was defaulting to 200K
91+
{"accounts/fireworks/models/glm-5p2", 1_048_576}, // alias -> GLM 5.3 // was defaulting to 200K
8892
{"accounts/fireworks/models/kimi-k3", 1_000_000}, // k3 ≠ the kimi-k2 262K case
8993
{"accounts/fireworks/models/kimi-k3", 1_000_000},
9094
{"accounts/fireworks/models/qwen3p8-max", 262_144},
9195
{"accounts/fireworks/models/qwen3p8-2p4t-a95b", 262_144}, // legacy path, still deployed
9296
{"accounts/fireworks/models/glm-5p3", 1_048_576},
9397
{"accounts/fireworks/models/glm-5p3-flash", 1_048_576},
9498
{"accounts/fireworks/models/deepseek-v4p1-flash", 1_048_576},
95-
{"accounts/fireworks/models/deepseek-v4-pro-0813", 1_048_576},
99+
{"accounts/fireworks/models/deepseek-v4-pro-0813", 1_048_576}, // alias -> V4.1 Flash
96100
{"accounts/fireworks/models/deepseek-v4-flash-0731", 1_048_576},
97101
{"gemini-3.1-pro-preview", 1_000_000},
98102
{"gemini-3.8-flash", 1_000_000},
@@ -115,15 +119,15 @@ func TestModelPricingCacheRates(t *testing.T) {
115119
{"grok-4.6", 0.5, 2.5},
116120
{"claude-sonnet-5", 0.2, 2.5},
117121
{"gpt-5.6-terra", 0.2, 2.5},
118-
{"accounts/fireworks/models/glm-5p2", 0.14, 1.75},
122+
{"accounts/fireworks/models/glm-5p2", 0.26, 1.75}, // alias -> GLM 5.3 cache rate
119123
{"accounts/fireworks/models/kimi-k3", 0.30, 3.75},
120124
// Qwen 3.8 Max publishes $0.25 cached, NOT the 0.1x default this used to inherit.
121125
{"accounts/fireworks/models/qwen3p8-max", 0.25, 2.5},
122126
{"accounts/fireworks/models/qwen3p8-2p4t-a95b", 0.25, 2.5}, // legacy path prices the same
123127
{"accounts/fireworks/models/glm-5p3", 0.26, 1.75},
124128
{"accounts/fireworks/models/glm-5p3-flash", 0.03, 0.1875},
125129
{"accounts/fireworks/models/deepseek-v4p1-flash", 0.007, 0.275},
126-
{"accounts/fireworks/models/deepseek-v4-pro-0813", 0.044, 1.32 * 1.25},
130+
{"accounts/fireworks/models/deepseek-v4-pro-0813", 0.007, 0.275}, // alias -> V4.1 Flash
127131
{"accounts/fireworks/models/deepseek-v4-flash-0731", 0.007, 0.275},
128132
{"gpt-image-2", 0.8, 10},
129133
}
@@ -272,3 +276,37 @@ func TestMinReasoningEffort(t *testing.T) {
272276
t.Errorf("unknown floor = %q, want \"\"", got)
273277
}
274278
}
279+
280+
// Aliases exist so a vendor decommission migrates pins instead of breaking them:
281+
// the retired name must resolve to the replacement entry, and a live label must
282+
// always win over an alias.
283+
func TestRetiredModelAliases(t *testing.T) {
284+
for _, c := range []struct{ alias, wantID string }{
285+
{"deepseek-v4-pro", "accounts/fireworks/models/deepseek-v4p1-flash"},
286+
{"accounts/fireworks/models/deepseek-v4-pro-0813", "accounts/fireworks/models/deepseek-v4p1-flash"},
287+
{"accounts/fireworks/models/deepseek-v4-flash-0731", "accounts/fireworks/models/deepseek-v4p1-flash"},
288+
{"accounts/fireworks/models/deepseek-v4-flash-vision-exp", "accounts/fireworks/models/deepseek-v4p1-flash"},
289+
{"glm-5p2", "accounts/fireworks/models/glm-5p3"},
290+
{"accounts/fireworks/models/glm-5p2", "accounts/fireworks/models/glm-5p3"},
291+
} {
292+
m, ok := LookupModel(c.alias)
293+
if !ok || m.ID != c.wantID {
294+
t.Errorf("LookupModel(%q) = (%q, %v), want %q", c.alias, m.ID, ok, c.wantID)
295+
}
296+
}
297+
// A live label resolves to itself, never to an alias entry.
298+
if m, ok := LookupModel("glm-5p3"); !ok || m.ID != "accounts/fireworks/models/glm-5p3" {
299+
t.Errorf("live label glm-5p3 = (%q, %v)", m.ID, ok)
300+
}
301+
// No alias may name a model that is still live — mustLoadModelCatalog panics
302+
// on that, so this just pins the invariant for readers.
303+
for _, m := range CatalogModels() {
304+
for _, a := range m.Aliases {
305+
for _, other := range CatalogModels() {
306+
if other.ID == a || (other.Label != "" && other.Label == a) {
307+
t.Errorf("alias %q on %q collides with live model %q", a, m.ID, other.ID)
308+
}
309+
}
310+
}
311+
}
312+
}

‎internal/agent/runtime/evict_turnstart_test.go‎

Lines changed: 5 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,7 @@
11
package runtime
22

33
import (
4+
"github.com/memcode-ai/memcode/catalog"
45
"io"
56
"strings"
67
"testing"
@@ -84,8 +85,10 @@ func TestCompactBudgetFollowsWindow(t *testing.T) {
8485
t.Setenv("MEMCODE_CONTEXT_SOFT_CAP", "")
8586
s := &Session{out: io.Discard, turn: newTurnState(), planCtl: &plan.Controller{}, model: "glm-5p2"}
8687

87-
// Nothing learned → the MODEL's window governs (glm-5p2 = 1M → 800K).
88-
if got, want := s.compactBudget(), 1_000_000*windowFallbackPct/100; got != want {
88+
// Nothing learned → the MODEL's window governs. Derived from the catalog, not
89+
// typed as a constant: a hardcoded window is the same absolute-number trap this
90+
// test exists to prevent, and it broke when glm-5p2 became an alias for 5.3.
91+
if got, want := s.compactBudget(), catalog.ContextWindow(s.model)*windowFallbackPct/100; got != want {
8992
t.Fatalf("pre-learning budget = %d, want %d (window-relative)", got, want)
9093
}
9194
// Learned 1M lane (960K usable) → the lane governs: 816K, no built-in clip.

0 commit comments

Comments
 (0)