Skip to content

Commit 9fc48f1

Browse files
github-actions[bot]erin2722claude
authored
chore: remove deprecated models from the catalog (#986)
## Model deprecation audit Each model below returned a definitive not-found / deprecated response from its provider (a single 404 / not-found is treated as authoritative). Account/region-scoped providers (bedrock, vertex, databricks) are surfaced for manual review, not auto-removed. ```json { "deprecations": [ { "model": "moonshotai/Kimi-K2.5", "provider": "baseten", "reason": "absent from list + probe not-found", "status": 410, "detail": "{\"error\":\"the model version you are trying to access has been deprecated.\"}\n" }, { "model": "nvidia/Nemotron-120B-A12B", "provider": "baseten", "reason": "absent from list + probe not-found", "status": 410, "detail": "{\"error\":\"the model version you are trying to access has been deprecated.\"}\n" }, { "model": "zai-org/GLM-5", "provider": "baseten", "reason": "absent from list + probe not-found", "status": 410, "detail": "{\"error\":\"the model version you are trying to access has been deprecated.\"}\n" }, { "model": "zai-org/GLM-5.1", "provider": "baseten", "reason": "absent from list + probe not-found", "status": 410, "detail": "{\"error\":\"the model version you are trying to access has been deprecated.\"}\n" }, { "model": "meta-llama/llama-4-scout-17b-16e-instruct", "provider": "groq", "reason": "absent from list + probe not-found", "status": 404, "detail": "{\"error\":{\"message\":\"The model `meta-llama/llama-4-scout-17b-16e-instruct` does not exist or you do not have access to it.\",\"type\":\"invalid_request_error\",\"code" }, { "model": "qwen/qwen3-32b", "provider": "groq", "reason": "absent from list + probe not-found", "status": 404, "detail": "{\"error\":{\"message\":\"The model `qwen/qwen3-32b` does not exist or you do not have access to it.\",\"type\":\"invalid_request_error\",\"code\":\"model_not_found\"}}\n" } ], "reportOnly": [ { "provider": "bedrock", "modelCount": 122 }, { "provider": "databricks", "modelCount": 36 }, { "provider": "fireworks", "modelCount": 266 }, { "provider": "vertex", "modelCount": 122 } ], "skipped": [ { "provider": "cohere", "reason": "no adapter" }, { "provider": "lepton", "reason": "no adapter" }, { "provider": "openrouter", "reason": "no adapter" }, { "provider": "replicate", "reason": "no adapter" } ] } ``` --------- Co-authored-by: erin2722 <16248113+erin2722@users.noreply.github.com> Co-authored-by: Erin McNulty <erin.mcnulty@braintrustdata.com> Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
1 parent 8a2ed7a commit 9fc48f1

3 files changed

Lines changed: 8 additions & 51 deletions

File tree

packages/proxy/schema/index.ts

Lines changed: 3 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -378,7 +378,6 @@ export const AvailableEndpointTypes: { [name: string]: ModelEndpointType[] } = {
378378
"open-mistral-nemo-2407": ["mistral"],
379379
mistral: ["ollama"],
380380
phi: ["ollama"],
381-
"meta-llama/llama-4-scout-17b-16e-instruct": ["groq"],
382381
"llama-3.3-70b-versatile": ["groq"],
383382
"llama-3.1-8b-instant": ["groq"],
384383
"accounts/fireworks/models/llama-guard-3-8b": ["fireworks"],
@@ -387,7 +386,6 @@ export const AvailableEndpointTypes: { [name: string]: ModelEndpointType[] } = {
387386
"accounts/fireworks/models/llama-guard-2-8b": ["fireworks"],
388387
"accounts/fireworks/models/deepseek-r1-distill-qwen-32b": ["fireworks"],
389388
"accounts/fireworks/models/deepseek-r1-distill-qwen-14b": ["fireworks"],
390-
"qwen/qwen3-32b": ["groq"],
391389
"llama3-3-70b": ["lepton"],
392390
"llama3-2-3b": ["lepton"],
393391
"llama3-2-1b": ["lepton"],
@@ -687,7 +685,7 @@ export const AvailableEndpointTypes: { [name: string]: ModelEndpointType[] } = {
687685
"deepcogito/cogito-v2-1-671b": ["together"],
688686
"essentialai/rnj-1-instruct": ["together"],
689687
"Qwen/Qwen3.6-Plus": ["together"],
690-
"zai-org/GLM-5.1": ["baseten", "together"],
688+
"zai-org/GLM-5.1": ["together"],
691689
"zai-org/GLM-5.2": ["baseten", "together"],
692690
"MiniMaxAI/MiniMax-M2.7": ["together"],
693691
"MiniMaxAI/MiniMax-M3": ["together"],
@@ -757,7 +755,7 @@ export const AvailableEndpointTypes: { [name: string]: ModelEndpointType[] } = {
757755
"accounts/fireworks/models/flux-1-schnell": ["fireworks"],
758756
"accounts/fireworks/models/flux-kontext-max": ["fireworks"],
759757
"accounts/fireworks/models/flux-kontext-pro": ["fireworks"],
760-
"zai-org/GLM-5": ["baseten", "together"],
758+
"zai-org/GLM-5": ["together"],
761759
"accounts/fireworks/models/glm-4p7": ["fireworks"],
762760
"zai-org/GLM-4.7": ["baseten"],
763761
"accounts/fireworks/models/glm-4p6": ["fireworks"],
@@ -920,11 +918,10 @@ export const AvailableEndpointTypes: { [name: string]: ModelEndpointType[] } = {
920918
"whisper-1": ["openai", "azure"],
921919
"babbage-002": ["openai", "azure"],
922920
"davinci-002": ["openai", "azure"],
923-
"nvidia/Nemotron-120B-A12B": ["baseten"],
924921
"fireworks_ai/WhereIsAI/UAE-Large-V1": ["fireworks"],
925922
"accounts/fireworks/models/kimi-k2p5": ["fireworks"],
926923
"accounts/fireworks/models/kimi-k2p6": ["fireworks"],
927-
"moonshotai/Kimi-K2.5": ["baseten", "together"],
924+
"moonshotai/Kimi-K2.5": ["together"],
928925
"moonshotai/Kimi-K2.6": ["baseten", "together"],
929926
"fireworks_ai/nomic-ai/nomic-embed-text-v1": ["fireworks"],
930927
"fireworks_ai/nomic-ai/nomic-embed-text-v1.5": ["fireworks"],

packages/proxy/schema/model_list.json

Lines changed: 2 additions & 45 deletions
Original file line numberDiff line numberDiff line change
@@ -6008,20 +6008,6 @@
60086008
"groq"
60096009
]
60106010
},
6011-
"meta-llama/llama-4-scout-17b-16e-instruct": {
6012-
"format": "openai",
6013-
"flavor": "chat",
6014-
"input_cost_per_mil_tokens": 0.11,
6015-
"output_cost_per_mil_tokens": 0.34,
6016-
"displayName": "Llama 4 Scout (17Bx16E)",
6017-
"experimental": true,
6018-
"deprecation_date": "2026-07-17",
6019-
"max_input_tokens": 131072,
6020-
"max_output_tokens": 8192,
6021-
"available_providers": [
6022-
"groq"
6023-
]
6024-
},
60256011
"llama-3.3-70b-versatile": {
60266012
"format": "openai",
60276013
"flavor": "chat",
@@ -6060,21 +6046,6 @@
60606046
"groq"
60616047
]
60626048
},
6063-
"qwen/qwen3-32b": {
6064-
"format": "openai",
6065-
"flavor": "chat",
6066-
"input_cost_per_mil_tokens": 0.29,
6067-
"output_cost_per_mil_tokens": 0.59,
6068-
"displayName": "Qwen3-32B",
6069-
"reasoning": true,
6070-
"experimental": true,
6071-
"deprecation_date": "2026-07-17",
6072-
"max_input_tokens": 131072,
6073-
"max_output_tokens": 40960,
6074-
"available_providers": [
6075-
"groq"
6076-
]
6077-
},
60786049
"groq/compound": {
60796050
"format": "openai",
60806051
"flavor": "chat",
@@ -8128,11 +8099,10 @@
81288099
"reasoning_budget": true,
81298100
"fallback_models": [
81308101
"anthropic.claude-opus-5",
8102+
"eu.anthropic.claude-opus-5",
81318103
"global.anthropic.claude-opus-5",
81328104
"publishers/anthropic/models/claude-opus-5",
8133-
"us.anthropic.claude-opus-5",
8134-
"eu.anthropic.claude-opus-5",
8135-
"au.anthropic.claude-opus-5"
8105+
"us.anthropic.claude-opus-5"
81368106
],
81378107
"max_input_tokens": 1000000,
81388108
"max_output_tokens": 128000,
@@ -10662,7 +10632,6 @@
1066210632
"max_input_tokens": 202752,
1066310633
"max_output_tokens": 128000,
1066410634
"available_providers": [
10665-
"baseten",
1066610635
"together"
1066710636
]
1066810637
},
@@ -10673,7 +10642,6 @@
1067310642
"output_cost_per_mil_tokens": 3.15,
1067410643
"input_cache_read_cost_per_mil_tokens": 0.2,
1067510644
"available_providers": [
10676-
"baseten",
1067710645
"together"
1067810646
]
1067910647
},
@@ -10706,20 +10674,9 @@
1070610674
"output_cost_per_mil_tokens": 3,
1070710675
"input_cache_read_cost_per_mil_tokens": 0.12,
1070810676
"available_providers": [
10709-
"baseten",
1071010677
"together"
1071110678
]
1071210679
},
10713-
"nvidia/Nemotron-120B-A12B": {
10714-
"format": "openai",
10715-
"flavor": "chat",
10716-
"input_cost_per_mil_tokens": 0.3,
10717-
"output_cost_per_mil_tokens": 0.75,
10718-
"input_cache_read_cost_per_mil_tokens": 0.06,
10719-
"available_providers": [
10720-
"baseten"
10721-
]
10722-
},
1072310680
"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
1072410681
"format": "openai",
1072510682
"flavor": "chat",

packages/proxy/scripts/deprecated_model_ids.json

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -76,6 +76,7 @@
7676
"magistral-small-2506",
7777
"meta-llama/Meta-Llama-3-8B-Instruct-Turbo",
7878
"meta-llama/llama-4-maverick-17b-128e-instruct",
79+
"meta-llama/llama-4-scout-17b-16e-instruct",
7980
"mistral-large-2411",
8081
"mistral-medium-3-5-26-04",
8182
"mistral-saba-24b",
@@ -85,6 +86,7 @@
8586
"moonshotai/Kimi-K2-Instruct-0905",
8687
"moonshotai/Kimi-K2-Thinking",
8788
"moonshotai/kimi-k2-instruct-0905",
89+
"nvidia/Nemotron-120B-A12B",
8890
"o1-mini",
8991
"o1-mini-2024-09-12",
9092
"o1-preview",
@@ -96,6 +98,7 @@
9698
"qwen-2.5-32b",
9799
"qwen-2.5-coder-32b",
98100
"qwen-qwq-32b",
101+
"qwen/qwen3-32b",
99102
"r1-1776",
100103
"sonar-reasoning",
101104
"zai-org/GLM-4.6"

0 commit comments

Comments
 (0)