Compare commits

...

245 Commits

Author SHA1 Message Date
Aiden Cline 89b834086a refactor: move sync implementation into core src 2026-05-21 18:06:25 -05:00
Aiden Cline 1ab2ff8163 Merge pull request #1826 from smakosh/add-llmgateway-models
feat: add new LLM Gateway text models
2026-05-21 17:58:29 -05:00
Frank 9468676683 update zen models 2026-05-21 18:42:36 -04:00
Claude 6cdd2f054b Merge upstream/dev into add-llmgateway-models; resolve gemini-3.5-flash conflict
# Conflicts:
#	providers/google/models/gemini-3.5-flash.toml
2026-05-21 21:48:48 +00:00
Aiden Cline b13abc9141 Merge pull request #1827 from anomalyco/update-xai-pricing
fix xAI long-context pricing
2026-05-21 16:45:43 -05:00
Aiden Cline e5ba264751 fix xAI long-context pricing 2026-05-21 16:41:36 -05:00
smakosh a7811fb522 refactor: use extends for gemini and qwen models
Add canonical google/gemini-3.5-flash and alibaba/qwen3.7-max defs and
have the llmgateway entries extend them, per PR review.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-21 23:38:57 +02:00
smakosh 605fae75d9 feat: add new LLM Gateway text models
Add Grok 4.20 (reasoning/non-reasoning), Gemini 3.5 Flash, and Qwen3.7 Max to the llmgateway provider.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-21 23:04:47 +02:00
Aiden Cline 26b05268ae Merge pull request #1824 from anomalyco/fix/vercel-gemini-35-flash
Add new Vercel AI Gateway models
2026-05-21 15:30:35 -05:00
Aiden Cline 1aee13d2e5 Add new Vercel AI Gateway models 2026-05-21 13:21:25 -05:00
Aiden Cline 0a924e6bb2 Merge pull request #1822 from anomalyco/automation/sync-models-xai
chore(sync): update xAI model catalog
2026-05-21 13:05:20 -05:00
Aiden Cline 9769b2b11d Merge pull request #1823 from anomalyco/automation/sync-models-openrouter
chore(sync): update OpenRouter model catalog
2026-05-21 13:05:13 -05:00
github-actions[bot] 1b0db099cf chore(sync): update OpenRouter model catalog 2026-05-21 17:56:13 +00:00
github-actions[bot] 16ed78587c chore(sync): update xAI model catalog 2026-05-21 17:56:11 +00:00
Frank 0b88965165 update zen models 2026-05-21 13:41:55 -04:00
Aiden Cline acc704ce39 Merge pull request #1797 from arnavchachra/add-crof-provider
add crof.ai provider with 21 models
2026-05-21 11:43:02 -05:00
Aiden Cline 51ad3b264e Merge pull request #1821 from anomalyco/sync-provider-ci
chore: automate provider sync jobs
2026-05-21 11:23:49 -05:00
Aiden Cline 146b6c7084 Merge pull request #1819 from Inceptron-Software/add_inceptron_provider
Add Inceptron provider
2026-05-21 11:21:27 -05:00
Aiden Cline 0e3cbe3c64 chore: automate provider sync jobs 2026-05-21 11:21:12 -05:00
Aiden Cline 604d4d66a4 Merge pull request #1820 from Suat-B/codex/xpersona-image-input-20260521
Add image input modality to Xpersona model
2026-05-21 11:00:49 -05:00
SuatB f5090028b8 Add image input modality to Xpersona model 2026-05-21 09:46:28 -05:00
Frank 4bad8faf29 update zen models 2026-05-21 09:05:13 -04:00
Oskar Gustafsson 0df2ccf586 Add Inceptron provider 2026-05-21 09:24:43 +02:00
Aiden Cline bafdc00b45 Merge pull request #1812 from anomalyco/openrouter-extends-sync
Sync OpenRouter models with extends
2026-05-20 21:07:21 -05:00
Aiden Cline 49840c013b Merge pull request #1814 from neonn0d/feat/stepfun-ai
feat(stepfun-ai): add international StepFun platform
2026-05-20 20:58:37 -05:00
Aiden Cline eccae0b54e sync openrouter models with extends 2026-05-20 20:32:11 -05:00
Aiden Cline 4cca29405f Merge pull request #1817 from dpuyosa/dev
Venice: Remove Grok 4.1 Fast and add Grok Build 0.1
2026-05-20 20:26:45 -05:00
Aiden Cline e40d9dd338 Merge pull request #1818 from anomalyco/cloudflare-sync-env
chore(sync): isolate cloudflare credentials
2026-05-20 20:26:22 -05:00
Aiden Cline 6a74991397 chore(sync): isolate cloudflare credentials 2026-05-20 20:19:47 -05:00
dpuyosa 035999cb58 [venice] Replace Grok 4.1 Fast with Grok Build 0.1
- Remove deprecated grok-41-fast model entry
- Add grok-build-0-1 with 200K token tiered pricing
- Update context to 256K and output limit to 65,536
2026-05-21 02:36:19 +02:00
Frank cec56bf1bc update zen models 2026-05-20 19:43:25 -04:00
Aiden Cline 85f0cdcb2f Merge pull request #1816 from anomalyco/xai-sync
Add PDF input modality to Grok models
2026-05-20 18:12:47 -05:00
Aiden Cline ef80d4df4e Infer PDF modality for xAI image models 2026-05-20 18:12:11 -05:00
Aiden Cline af0ef00109 Update xAI Grok PDF modalities 2026-05-20 18:06:33 -05:00
Aiden Cline 5fdcea6b36 Merge pull request #1815 from anomalyco/cloudflare-ai-gateway
chore(sync): add cloudflare workers ai sync
2026-05-20 18:02:28 -05:00
Aiden Cline 31e56480b4 chore(sync): add cloudflare workers ai sync 2026-05-20 16:55:29 -05:00
Aiden Cline 92a621594e Merge pull request #1813 from anomalyco/sync-xai
chore(sync): add xai model sync
2026-05-20 16:02:38 -05:00
Aiden Cline d2db353ceb chore: ignore sync reports 2026-05-20 16:01:51 -05:00
Aiden Cline 900ae509d2 Merge pull request #1808 from ajussak/scaleway
Added Mistral Medium 3.5 128B from Scaleway
2026-05-20 15:58:00 -05:00
neo 9d60164243 feat(stepfun-ai): add international StepFun platform
StepFun runs two separate platforms with distinct accounts/keys:
platform.stepfun.com (China, already covered by providers/stepfun) and
platform.stepfun.ai (international). Keys are not interchangeable
across the two — .ai keys are rejected by api.stepfun.com as
invalid_api_key.

Stepfun's own opencode integration guide instructs users to point at
https://api.stepfun.ai/step_plan/v1. This adds providers/stepfun-ai
for that endpoint, symlinking the shared chat models. Follows the
moonshotai / moonshotai-cn pattern.
2026-05-20 20:43:47 +02:00
Adrien Jussak cd3e99025f Update Mistral Medium 3.5 128B model configuration to extend from mistral-medium-2604 and adjust context window size. 2026-05-20 20:43:45 +02:00
Aiden Cline 1098981eb6 chore(sync): add xai model sync 2026-05-20 13:23:02 -05:00
Aiden Cline 27a151cf53 Merge pull request #1811 from anomalyco/xai-grok-build-model
Add xAI Grok Build model
2026-05-20 13:05:06 -05:00
Aiden Cline 41ff42ab7f Merge pull request #1810 from fhennerkes/dev
poe: add Gemini-3.5-Flash model
2026-05-20 13:04:47 -05:00
Aiden Cline adf1cbdecd add xai grok build model 2026-05-20 13:04:13 -05:00
Frank 10ddc78ce0 update zen models 2026-05-20 14:02:01 -04:00
fhennerkes 7ae897e440 poe: add Gemini-3.5-Flash model
Add new Google model from Poe API (released 2026-05-19).
Uses extends format inheriting from google/gemini-3.5-flash with
Poe-specific overrides (name format, no temperature, markup pricing,
limited input modalities).

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-20 10:52:15 -07:00
Aiden Cline 02e452c2e8 Merge pull request #1805 from anomalyco/sync-google
sync google models
2026-05-20 10:53:57 -05:00
Aiden Cline f05b63fff5 Merge pull request #1807 from anomalyco/automation/sync-models-aggregators
chore(sync): update aggregator model catalogs
2026-05-20 10:34:30 -05:00
Adrien Jussak e277d60236 Add Mistral Medium 3.5 128B to Scaleway 2026-05-20 16:14:37 +02:00
github-actions[bot] b01277a737 chore(sync): update aggregator model catalogs 2026-05-20 09:25:37 +00:00
Frank a5da5aa429 update zen models 2026-05-20 04:14:28 -04:00
Aiden Cline 5ceac8a58b updates 2026-05-19 23:43:01 -05:00
Aiden Cline 11e1d5623a Merge pull request #1806 from Cahl-Dee/grid-model-updates-2026-05
Grid model updates 2026-05
2026-05-19 23:42:26 -05:00
Aiden Cline f107afc57c sync 2026-05-19 22:56:50 -05:00
Carl DiClementi e789d7c1f2 Merge branch 'anomalyco:dev' into grid-model-updates-2026-05 2026-05-19 16:06:23 -05:00
Cahl-Dee 899668ad49 added new code and agent models and updated existing text models 2026-05-19 16:04:58 -05:00
Aiden Cline 8c677f0134 sync google models 2026-05-19 15:58:06 -05:00
Aiden Cline 462c7877d9 add gemini 3.5 flash 2026-05-19 15:43:38 -05:00
Aiden Cline 55871f9dca Merge pull request #1804 from Adanlink/dev
feat: add deepseek-v4-flash to the fireworks-ai provider
2026-05-19 15:29:33 -05:00
Aiden Cline 4f7194a3c8 test 2026-05-19 15:09:11 -05:00
Aiden Cline f65f0148da add sync guide 2026-05-19 15:08:44 -05:00
Adán 14952f8855 Rename deepseek-v4-flash to deepseek-v4-flash.toml 2026-05-19 18:56:43 +02:00
Adán d7c6d3ad12 Add deepseek-v4-flash model configuration 2026-05-19 18:53:48 +02:00
Aiden Cline 356bc79d08 Merge pull request #1637 from elvexai/fix/amazon-bedrock-kimi-token-limits
fix: Token limits for Amazon Bedrock Kimi K2 models
2026-05-19 09:42:30 -05:00
Aiden Cline a89b1ed726 Merge pull request #1801 from bas3line/sync-routing-run-models
Sync routing.run model catalog
2026-05-19 09:41:38 -05:00
bas3line a998576773 fix(routing-run): match live model metadata 2026-05-19 10:38:58 +05:30
bas3line fbe842bbea fix(routing-run): expose reasoning metadata 2026-05-19 08:39:51 +05:30
bas3line 6c0c3d1b10 fix(routing-run): sync model catalog 2026-05-19 07:37:53 +05:30
Aiden Cline db0a7cf611 Merge pull request #1798 from anomalyco/rework-sync-logic
sync: centralize aggregator model updates
2026-05-18 20:12:50 -05:00
Aiden Cline d775e37e3b Merge pull request #1800 from jerome-benoit/feat/sap-ai-core-gpt-5.4
feat(sap-ai-core): add GPT-5.4
2026-05-18 20:12:21 -05:00
Jérôme Benoit 36753063d9 feat(sap-ai-core): add GPT-5.4
Add gpt-5.4 with availability date from official SAP source.

Drop [[cost.tiers]] from gemini-2.5-pro pending SAP-side tiered
pricing confirmation; sap-ai-core now declares no per-model tiers
(SAP Note 3437766 is login-gated and authoritative for capacity
unit conversion rates).
2026-05-19 02:58:29 +02:00
Aiden Cline 5ee955297a sync: drop vercel catalog updates 2026-05-18 19:07:14 -05:00
Aiden Cline 1b77511903 Merge pull request #1799 from vglafirov/add-gitlab-gpt-5-5
feat(gitlab): add Agentic Chat (GPT-5.5) model
2026-05-18 15:24:18 -05:00
Aiden Cline 8896ead7bf sync: fix vercel pricing tiers 2026-05-18 14:52:40 -05:00
Vladimir Glafirov eb96594d47 feat(gitlab): add Agentic Chat (GPT-5.5) model
Adds duo-chat-gpt-5-5 to the GitLab provider. The GitLab AI Gateway
proxies this model to OpenAI's gpt-5.5-2026-04-23 backend with a
1.05M token context window (922k input + 128k output).

Source: gitlab-org/modelops/applied-ml/code-suggestions/ai-assist
models.yml (gpt_5_5 entry with proxy_provider: openai).

The gitlab-ai-provider npm package exposes this model id starting in
v6.7.0.
2026-05-18 20:44:34 +02:00
Aiden Cline 327332efe3 Merge pull request #1794 from bas3line/add-routing-run-provider
Add routing.run provider
2026-05-18 12:32:34 -05:00
Aiden Cline 5020951745 sync: refresh openrouter after dev merge 2026-05-18 12:30:29 -05:00
Aiden Cline cb6f97774e Merge remote-tracking branch 'origin/dev' into rework-sync-logic 2026-05-18 12:29:28 -05:00
Aiden Cline 7f8b493b0c Merge pull request #1795 from delafthi/delafthi/lxxqxzktnozv
fix(providers/novita-ai): use lowercase model names
2026-05-18 12:28:32 -05:00
Aiden Cline d65a862533 sync: centralize aggregator model updates 2026-05-18 12:12:15 -05:00
arnavchachra 9420048dfe fix crof model limits and reasoning flag to match Crof API 2026-05-18 21:51:19 +05:30
arnavchachra 8db6c27634 add crof provider with 21 models 2026-05-18 21:40:11 +05:30
Victor Navarro 8e710e19ea bring back old bick-pickle
Added interleaved section with reasoning_content field and removed provider section.
2026-05-18 11:44:14 +02:00
Frank 36c6896e97 update zen models 2026-05-17 22:58:06 -04:00
Aiden Cline a8be548a5d Merge pull request #1416 from Luew2/add-lilac-provider
Add Lilac provider
2026-05-17 19:23:07 -05:00
Luew2 8c2fae4ab0 Keep exact Lilac Gemma model name 2026-05-17 17:19:53 -07:00
Luew2 5e7fad350d Align Lilac Gemma display name 2026-05-17 17:19:04 -07:00
Luew2 feb85ef2c9 Align Lilac provider with registry conventions 2026-05-17 17:15:40 -07:00
Luew2 d4161ebf24 Follow models.dev conventions for Lilac provider 2026-05-17 17:11:03 -07:00
Luew2 b91ab02e2b Add Lilac MiniMax M2.7 model 2026-05-17 17:07:44 -07:00
Luew2 dd09d07f75 Update Lilac Kimi model to K2.6 2026-05-17 17:07:44 -07:00
Luew2 4515f85d47 Add Lilac cache pricing 2026-05-17 17:07:44 -07:00
Luew2 97572240e1 Add Gemma 4 31B IT model
Adds google/gemma-4-31b-it to the Lilac provider ($0.11/M input,
$0.35/M output, 262K context, native multimodal with image/video).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-17 17:07:44 -07:00
Luew2 80cc04e9e5 Fix Kimi K2.5 output limit to 262,144 tokens 2026-05-17 17:07:44 -07:00
Luew2 d532ebb89b Use purple gradient for Lilac logo (brand colors #6451dc → #b6a6f9) 2026-05-17 17:07:44 -07:00
Luew2 9bae3e887f Replace placeholder logo with Lilac icon mark (currentColor) 2026-05-17 17:07:44 -07:00
Luew2 0be69bf872 Add Lilac provider
Add Lilac as an OpenAI-compatible provider serving:
- z-ai/glm-5.1: Z.ai's flagship agentic model (754B MoE, 202.8K context)
- moonshotai/kimi-k2.5: Moonshot AI's multimodal reasoning model (1T MoE, 262K context)

API: https://api.getlilac.com/v1
Docs: https://docs.getlilac.com
2026-05-17 17:07:44 -07:00
Thierry Delafontaine 0ed38cbecf fix(providers/novita-ai): use lowercase model names
Mixed case naming causes conflicts on case-insensitive filesystems like macOS.
2026-05-17 21:17:46 +02:00
bas3line e65703382a feat: add routing.run provider 2026-05-17 20:27:56 +05:30
Aiden Cline 748754c99b Merge pull request #1792 from monotykamary/fix/neuralwatt-context-limits
fix(neuralwatt): sync context window and output limits with upstream API
2026-05-16 13:14:05 -05:00
Tom X Nguyen ec9c12d0fc fix(neuralwatt): sync context window and output limits with upstream API
Updates all 14 neuralwatt model TOML files with corrected context window
and max output token values as reported by the Neuralwatt API:

- Devstral-Small-2-24B-Instruct-2512: 262,144 -> 262,128
- GLM-5/GLM-5.1 variants: 200,000 -> 202,736
- GPT-OSS-20B: 16,384 -> 16,368
- Kimi-K2.5/K2.6 variants: 262,144 -> 262,128
- MiniMax-M2.5: 196,608 -> 196,592
- Qwen3.5-397B variants: 262,144 -> 262,128
- Qwen3.6-35B variants: 131,072 -> 131,056

Also fixes the README: moves kimi-k2.6-fast from 'Reasoning Models' to
'Fast Variants' and removes incorrect claim that fast variants support
reasoning.
2026-05-16 23:16:34 +07:00
Aiden Cline ac81822c89 Merge pull request #1777 from berget-ai/feat/berget-kimi-k2.6
feat: add Kimi K2.6 to berget.ai
2026-05-16 06:09:00 -05:00
Aiden Cline d32ed764bd Merge pull request #1791 from anomalyco/automation/sync-openrouter-models
Sync OpenRouter models
2026-05-16 06:08:09 -05:00
Christian Landgren 45fb951c42 feat: add Kimi K2.6 to berget.ai
Add Moonshot AI Kimi K2.6 model to berget.ai provider catalog.

- 262K context window
- 16K output tokens
- Text input/output
- Supports: reasoning, structured output, tool calling
- Pricing: /bin/zsh.83/M input, .85/M output (EUR-based)
- Open weights
2026-05-16 12:46:35 +02:00
github-actions[bot] 260d79b2d5 Sync OpenRouter models 2026-05-16 08:54:41 +00:00
Aiden Cline 746b9caf79 Merge pull request #1785 from jerome-benoit/feat/sap-ai-core-opus-4-7
feat(sap-ai-core): add Claude Opus 4.7 and sync model specs
2026-05-15 23:24:47 -05:00
Aiden Cline 8362b55503 Merge pull request #1786 from Ardakilic/chore/kilo-sync-20260516
providers(kilo): sync upstream
2026-05-15 23:24:34 -05:00
Aiden Cline dde3953a9f Merge pull request #1787 from Suat-B/codex/xpersona-www-api-url
Fix Xpersona API base URL
2026-05-15 23:24:06 -05:00
Aiden Cline 0a1695212c Merge pull request #1788 from Jaaneek/xai-may-15-2026-retirement
xai: drop models retired May 15, 2026 + add Grok Imagine models
2026-05-15 23:23:52 -05:00
Jaaneek 89fbb6bb69 xai: drop models retired May 15, 2026 + add Grok Imagine models 2026-05-16 01:57:15 +01:00
SuatB 005fe0fb5a Fix Xpersona provider API URL 2026-05-15 18:37:41 -05:00
Frank e283875ce7 update zen models 2026-05-15 17:24:53 -04:00
Arda Kilicdagi 3598019251 providers(kilo): sync upstream 2026-05-16 01:11:25 +04:00
Jérôme Benoit e9ad8b0a3f feat(sap-ai-core): add Claude Opus 4.7 and sync model specs 2026-05-15 22:16:07 +02:00
Aiden Cline 0ee78eeda5 sync: openrouter models 2026-05-15 10:20:36 -05:00
Aiden Cline 0a5b33e518 Merge pull request #1778 from zhenjunchen-png/add-orcarouter
feat: add OrcaRouter as a new provider
2026-05-15 10:16:29 -05:00
Aiden Cline 22416dda64 Merge pull request #1783 from anomalyco/sync-openrouter
add sync script for openrouter, sync openrouter models
2026-05-15 10:12:02 -05:00
Aiden Cline eba2702e3a fix: families 2026-05-15 10:03:39 -05:00
Aiden Cline c2c5cc8f21 add sync script for openrouter, sync openrouter models 2026-05-15 10:00:32 -05:00
zhenjun.chen 022b1b9946 feat(orcarouter): expand to 80 chat models and add brand logo
Adds 55 additional upstream-mirrored models alongside the existing 25,
covering the full OrcaRouter chat catalog as exposed by
https://www.orcarouter.ai/api/pricing (text-only chat — TTS, embeddings,
video, and image generation are filtered out).

Per-namespace upstream mappings used by [extends]:

  OrcaRouter ns  -> models.dev provider
  ---------------- + ---------------
  openai         -> openai
  anthropic      -> anthropic   (dot version -> dash, e.g. opus-4.7 -> opus-4-7)
  google         -> google
  deepseek       -> deepseek
  qwen           -> alibaba
  grok           -> xai
  kimi           -> moonshotai
  minimax        -> minimax     (minimax-m2.7 -> MiniMax-M2.7)
  z-ai           -> zai

OrcaRouter-specific aliases (dated snapshots like gpt-5-2025-08-07,
search-preview variants, qwen3-vl-* visual variants) are excluded from
v1 because their upstream canonical files do not yet exist in models.dev.

Also adds providers/orcarouter/logo.svg.
2026-05-15 14:48:31 +08:00
Aiden Cline 8269e04222 Merge pull request #1782 from Suat-B/codex/xpersona-limits-logo
Update Xpersona limits, cutoff, and logo
2026-05-15 00:29:32 -05:00
Aiden Cline a75cf2ed1c Merge pull request #1552 from aredridel/as/add-umans
feat(models): add umans.ai coding plan
2026-05-14 22:40:39 -05:00
Aria Stewart 7b00aafa79 feat(models): umans.ai definitions 2026-05-14 23:39:30 -04:00
Suat-B e54e0fc8b5 Update Xpersona limits, cutoff, and logo 2026-05-15 03:22:16 +00:00
Aiden Cline 38611e75fa Merge pull request #1781 from dpuyosa/feat/add-venice-claude-opus-4-7-fast-model
Venice: Add Claude Opus 4.7 Fast model
2026-05-14 22:08:14 -05:00
dpuyosa e62c1e973e [venice] Add Claude Opus 4.7 Fast model
- New pricing with 36/180 input/output per million tokens
- 1M context window with 128K output limit
- Text+image input, text-only output
2026-05-15 01:32:24 +02:00
Aiden Cline c2b3c601e4 Merge pull request #1724 from isaachuangGMICLOUD/feat/add-gmicloud-provider
providers(gmicloud): add GMI Cloud provider
2026-05-14 17:26:50 -05:00
Aiden Cline 9ff1d36a21 Merge pull request #1476 from Vect0rM/feat/add-atomic-chat-provider
feat: add Atomic Chat provider
2026-05-14 17:18:21 -05:00
Aiden Cline e0f4042ad1 Merge pull request #1776 from kapelame/docs/minimax-token-plan-rename
providers(minimax): rename Coding Plan → Token Plan in display labels
2026-05-14 17:16:28 -05:00
Frank 14736ba4b6 update zen models 2026-05-14 17:15:53 -04:00
Aiden Cline 9351d68731 Merge pull request #1772 from nearai/add-nearai
Add NEAR AI Cloud provider
2026-05-14 10:30:28 -05:00
Aiden Cline 7a5a1d2aff Merge pull request #1779 from NameIsHiki/siliconflow-deepseek-v4
feat(siliconflow): add DeepSeek v4 models
2026-05-14 10:29:52 -05:00
zhenjun.chen 699284ce91 chore(orcarouter): drop oversize logo, fall back to models.dev default
The previously committed logo is ~100KB; existing wrapper-provider logos
(openrouter, llmgateway, kilo, aihubmix, ambient) are all 0.3-6KB and use
`currentColor`. Falling back to the default logo per README:

  > If we don't have a provider's logo, a default logo is served instead.

A properly-sized currentColor logo will follow in a separate PR.
2026-05-14 21:37:48 +08:00
Hiki 21ce5c3ac8 Create deepseek-v4-flash.toml 2026-05-14 15:26:37 +02:00
Hiki 3485cf52d0 Create deepseek-v4-pro.toml 2026-05-14 15:22:53 +02:00
Frank 99e8f25c78 update zen models 2026-05-14 08:59:42 -04:00
zhenjun.chen 7102978cb4 feat: add OrcaRouter provider
OrcaRouter is an OpenAI-compatible meta-router aggregating 150+ LLMs
(OpenAI, Anthropic, Google, xAI, DeepSeek, Qwen, Kimi, MiniMax, ...)
behind a single API key, with a virtual orcarouter/auto smart-routing
entry that picks an upstream per request.

This initial scope covers 26 models (1 AUTO router + 25 upstream mirrors
using [extends]). Pricing computed from https://www.orcarouter.ai/api/pricing
on 2026-05-14: input = model_ratio * $2, output = model_ratio *
completion_ratio * $2 (USD per 1M tokens).

Disclosure: I'm an engineer on the OrcaRouter team.
2026-05-14 20:55:47 +08:00
kapelame 530f60c69c providers(minimax): rename Coding Plan → Token Plan in display labels
The product was renamed from "Coding Plan" to "Token Plan" when its
scope expanded beyond coding to cover all MiniMax modalities (text,
speech, video, music, image). Per
https://platform.minimax.io/docs/token-plan/intro:
"Token Plan extends upon our former Coding Plan."

Updates display name and doc URL for the two affected provider
catalog entries. Provider IDs (minimax-coding-plan,
minimax-cn-coding-plan) are intentionally unchanged for backward
compatibility — anyone with these IDs in opencode.json or
elsewhere keeps working. Old /coding-plan/* URLs still 307-redirect
to the new /token-plan/* paths upstream.

Region disambiguation stays as the URL in parens (matching the
existing minimax / minimax-cn naming convention) — no "China" word
added, since the URL already conveys the region cleanly in the
provider picker.
2026-05-14 19:18:03 +08:00
Misha Skvortsov 2415c5be21 fix(atomic-chat): update logo.svg with new design
Replaces the existing logo.svg file with an updated design for the Atomic Chat provider. This change enhances the visual branding of the application.
2026-05-14 10:58:36 +03:00
Aiden Cline 85aba468cd Merge pull request #1767 from Suat-B/codex/xpersona-provider-20260513
Add Xpersona provider
2026-05-13 23:20:20 -05:00
Aiden Cline d1ec1ba777 Merge pull request #1773 from ambient-gregory/dev
feat: add Ambient provider with GLM-5.1 and Kimi K2.6
2026-05-13 19:20:20 -05:00
Gregory 0f94bf16ec fix(ambient): shrink logo display size to match other providers 2026-05-13 19:33:26 -04:00
Aiden Cline 506e8f48a9 Merge pull request #1770 from EriDeLee/dev
chore(aihubmix): sync model catalog
2026-05-13 17:43:30 -05:00
Aiden Cline 3480bc5992 Merge pull request #1775 from michaelnchin/fix/amazon-bedrock-gpt-oss-tokens
fix: Output tokens for Bedrock GPT-OSS models
2026-05-13 17:36:42 -05:00
Michael Chin fde97814ef fix: Output tokens for Bedrock GPT-OSS models 2026-05-13 14:45:17 -07:00
Gregory ff7eddcb70 feat: add Ambient provider with GLM-5.1 and Kimi K2.6
Adds the Ambient inference provider (api.ambient.xyz) with an initial
catalog of GLM-5.1 and Kimi K2.6, plus a generator script that pulls
from /v1/models so pricing and limits stay in sync with the upstream API.

Run `bun run ambient:generate` to refresh model TOMLs.
2026-05-13 11:30:45 -04:00
Evrard-Nil Daillet 5cbab85b8d Add nearai logo.svg from cloud.near.ai favicon 2026-05-13 16:32:08 +02:00
Evrard-Nil Daillet 6f9820de9f Add NEAR AI Cloud provider
Adds nearai as an OpenAI-compatible provider at https://cloud-api.near.ai/v1
serving 33 models. First-party mirrors (anthropic/openai/google) use `extends`;
NEAR-hosted open-weight models (Qwen, GLM-5.1-FP8, gpt-oss, whisper, FLUX) have
full definitions.

Pricing and context limits sourced from cloud-api.near.ai/v1/models.
2026-05-13 16:32:08 +02:00
EriDeLee cdfb429098 chore(aihubmix): sync model catalog 2026-05-13 21:55:32 +08:00
Victor Navarro 1c2546af8a perf: virtualize models table and other improvements 2026-05-13 12:29:27 +02:00
Suat-B 30b3e677fd Add Xpersona provider 2026-05-13 00:39:09 -05:00
Suat-B 3b37eee86e Add Xpersona provider 2026-05-13 00:39:08 -05:00
Suat-B 71f069670e Add Xpersona provider 2026-05-13 00:39:07 -05:00
Aiden Cline f401672689 Merge pull request #1766 from michaelnchin/fix/amazon-bedrock-structured-output-05122026-2
fix: add structured_output=True for more supported Bedrock models
2026-05-12 23:24:39 -05:00
Michael Chin 2a0d86a034 update structured_output for more Bedrock models 2026-05-12 20:52:36 -07:00
Aiden Cline d9439cdf2f Merge pull request #1762 from zxyaction/feat/add-auriko-provider
feat: add Auriko provider with 15 models
2026-05-12 22:16:02 -05:00
Aiden Cline 3d443d568d Merge pull request #1765 from michaelnchin/fix/amazon-bedrock-structured-output-05122026
fix: update structured_output for Bedrock Claude 4.x models
2026-05-12 22:15:51 -05:00
Michael Chin a76c8fe9dd fix: update structured_output for Bedrock Claude 4.x models 2026-05-12 20:07:12 -07:00
Aiden Cline d08e8d6cc1 Merge pull request #1763 from Tavernari/feat/add-claudinio-provider
feat: add Claudinio provider
2026-05-12 19:13:54 -05:00
Victor Carvalho Tavernari 4c06e44047 fix: use currentColor in logo SVG per contributing guidelines 2026-05-12 23:59:03 +01:00
Aiden Cline 5e344ded49 Merge pull request #1755 from anomalyco/correct-context-tracking
feat: add new context pricing tiers
2026-05-12 17:40:48 -05:00
Aiden Cline 458b7f4d1a use Venice context tier thresholds 2026-05-12 17:39:40 -05:00
Aiden Cline baf4432140 Merge pull request #1759 from NameIsHiki/deepinfra-xiaomi-mimo-models
feat(deepinfra): add Xiaomi MiMo v2.5 and v2.5 Pro
2026-05-12 17:10:59 -05:00
Aiden Cline bbf72ea4e0 Merge pull request #1764 from anomalyco/add-anthropic-opus-4-7-fast-mode
Add fast mode for Anthropic Opus 4.7
2026-05-12 17:10:49 -05:00
Aiden Cline 8f9adc7567 fix generated tier change detection 2026-05-12 17:01:35 -05:00
Aiden Cline addaf1c036 add fast mode for anthropic opus 4.7 2026-05-12 17:01:05 -05:00
Victor Carvalho Tavernari 55d16a58b6 feat: add claudinio provider (OpenAI-compatible, 256K ctx, $0.50/$2.00 per MTok) 2026-05-12 22:33:45 +01:00
Hiki 72a4deab66 Update mimo-v2.5.toml 2026-05-12 23:23:53 +02:00
Hiki 8dd829a187 Update mimo-v2.5-pro.toml 2026-05-12 23:23:01 +02:00
Aiden Cline a671cc05d5 align cost tiers with model schema 2026-05-12 16:01:49 -05:00
Aiden Cline 656c6f08a7 Merge pull request #1761 from Sewer56/deprecate-wafer-models
providers/wafer.ai: Remove DeepSeek-V4-Pro and MiniMax-M2.7 models
2026-05-12 15:51:10 -05:00
Aiden Cline 8979741a32 Merge pull request #1760 from Ardakilic/fix/kilo/kimik26
Fix: Kimi k2.6 definition on Kilo Gateway
2026-05-12 15:50:53 -05:00
Frank 82851b9a3d update zen models 2026-05-12 16:44:41 -04:00
zxy_action ae511892d7 feat: add Auriko provider with 15 models
All models use [extends] to inherit from canonical definitions,
overriding only Auriko-specific pricing. Omits remove cost tiers
and features Auriko doesn't carry.

Models: claude-opus-4-{6,7}, claude-sonnet-4-6, deepseek-v4-{pro,flash},
gemini-{2.5-pro,2.5-flash,3.1-pro-preview}, grok-4.3, kimi-k2.{5,6},
minimax-m2-7{,-highspeed}, glm-5.1, qwen-3.6-plus
2026-05-12 13:34:34 -07:00
Sewer56 9312418242 Changed: Remove DSv4 Pro & MiniMax M2.7 from models.dev 2026-05-12 20:16:08 +01:00
Arda Kılıçdağı 122627a852 fix: Kimi k2.6 definition on Kilo Gateway 2026-05-12 20:35:58 +03:00
Hiki 6883e793ce Create mimo-v2.5-pro.toml 2026-05-12 18:22:41 +02:00
Hiki e69064709b Update mimo-v2.5.toml 2026-05-12 18:19:52 +02:00
Hiki bd8e582b96 Create mimo-v2.5.toml 2026-05-12 18:03:09 +02:00
Shoubhit Dash 21945db90f Merge pull request #1662 from anomalyco/nxl/add-sarvam-provider
provider(sarvam): add chat models
2026-05-12 13:31:59 +05:30
Aiden Cline 1771e02be8 Merge pull request #1660 from Alex-wuhu/feat/novita-ai-sync-models
provider(novita-ai): sync latest models
2026-05-11 23:15:44 -05:00
Aiden Cline bb08fc26e9 Merge pull request #1706 from rohita5l/rohit/addDatabricks
Add Databricks as a provider
2026-05-11 19:28:34 -05:00
Aiden Cline 2e015de42d preserve generated tier thresholds 2026-05-11 17:07:19 -05:00
Rohit Agrawal 914a3d9d18 fix: restore bun.lock to use default registry instead of Databricks npm proxy
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-05-11 17:54:26 -04:00
Rohit Agrawal bab01dd9ab refactor: move databricks generate to packages/core/script following repo conventions
Addresses review feedback by removing AI SDK dependencies from package.json
and aligning with the Vercel/Helicone/Wandb pattern.

- Move generate-databricks.ts to packages/core/script/
- Add databricks:generate to root scripts
- Remove smoke test and runtime filtering (catalog should reflect what the
  upstream API exposes; AI SDK compatibility is a downstream concern)
- Add --dry-run and --new-only flags
- Merge with existing TOMLs instead of nuking them; warn about orphans
- Restore databricks-gemini-3-pro and databricks-gemini-3-1-pro
- Drop @ai-sdk/openai-compatible, ai, zod from root dependencies

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-05-11 17:54:04 -04:00
Rohit Agrawal bdaae956af feat: add AI SDK compatibility test to generate script, remove incompatible models
Generate script now smoke-tests each model with streamText after writing TOMLs
and removes any that return empty responses (incompatible with @ai-sdk/openai-compatible).
Removes databricks-gemini-3-pro and databricks-gemini-3-1-pro which return content
as array with thoughtSignature that the AI SDK cannot parse.

Also adds test-databricks.ts for standalone smoke testing.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-05-11 17:53:44 -04:00
Rohit Agrawal 6f118145c0 fix: inline gpt-oss model metadata instead of invalid extends path
openrouter models in subdirectories can't use extends (schema requires
provider/model format); resolve() now inlines the source TOML content directly.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-05-11 17:53:44 -04:00
Rohit Agrawal 386eaed119 add databricks 2026-05-11 17:53:44 -04:00
Aiden Cline 151e9c9071 fix tiered cost generation 2026-05-11 16:43:32 -05:00
Isaac ba99f1edce Add GMI Cloud GLM models 2026-05-11 14:40:03 -07:00
Aiden Cline b96b074a3b fix long-context cost tier omissions 2026-05-11 16:31:07 -05:00
Aiden Cline 2593e131a1 wip 2026-05-11 16:11:19 -05:00
Aiden Cline 4aebbe5ca3 Merge pull request #1754 from BruceMacD/brucemacd/fix-ollama-kimi-k2-6-model-id
fix ollama cloud kimi k2.6 model id
2026-05-11 14:57:03 -05:00
Bruce MacDonald b98495e9c8 fix ollama cloud kimi k2.6 model id 2026-05-11 12:40:17 -07:00
Frank bc95b42ccd update zen models 2026-05-11 11:26:08 -04:00
Aiden Cline 6139fb8c69 Merge pull request #1598 from mugnimaestra/feat/chutes-generate-script
feat(chutes): add API-driven model generator script
2026-05-11 09:30:38 -05:00
Aiden Cline 3070758007 Merge pull request #1678 from 5kahoisaac/chore/nvidia-models
Sync NVIDIA endpoint model catalog
2026-05-11 09:29:54 -05:00
Frank 5525e83de4 update zen models 2026-05-11 10:00:09 -04:00
Frank 359fd879b8 update zen models 2026-05-10 03:54:03 -04:00
Frank 01b5a1a656 update zen models 2026-05-10 02:52:44 -04:00
Frank b1958be099 update zen models 2026-05-10 02:42:44 -04:00
Aiden Cline 08aa068523 Temporarily remove kiro provider and models 2026-05-10 01:19:47 -05:00
Aiden Cline f31ad0b02f Merge pull request #1738 from mattiacerutti/chore/remove-gh-copilot-deprecated
chore(copilot): mark deprecated models
2026-05-09 15:42:19 -05:00
Aiden Cline 585aa7fa1b Merge pull request #1741 from EriDeLee/dev
Update aihubmix models
2026-05-09 15:42:03 -05:00
Aiden Cline c42a327b3e Merge pull request #1745 from mads-digitial-solutions/patch-1
Update Google provider docs url from pricing page to models page
2026-05-09 15:41:51 -05:00
mads-digitial-solutions 92ebbfb5c4 Update provider.toml
Update Google provider docs URL from the pricing page to the models page
2026-05-09 19:49:19 +01:00
Aiden Cline 535fe8c971 Merge pull request #1744 from OpeOginni/fix/bedrock-model-ids
chore(bedrock): Getting rid of legacy Amazon Bedrock model offerings
2026-05-09 13:48:52 -05:00
OpeOginni a3b4bfc16c fix(bedrock): remove uneeded model configurations 2026-05-09 20:35:34 +02:00
OpeOginni d0fcd6f11f fix(bedrock): align models with current docs 2026-05-09 20:29:11 +02:00
OpeOginni e55cd54218 fix(bedrock): remove legacy model entries 2026-05-09 20:21:33 +02:00
OpeOginni 0d73b82b9f fix(bedrock): restore regional model IDs 2026-05-09 20:18:59 +02:00
Aiden Cline 83c7e2b63f Merge pull request #1742 from Adam8234/add-firepass-provider
feat: add Fireworks (Firepass) provider
2026-05-09 12:45:09 -05:00
Adam 83ae4cf813 feat: add Fireworks (Firepass) provider
Adds the Fireworks AI Firepass subscription provider.
- Provider uses a dedicated FIREPASS_API_KEY
- Uses @ai-sdk/openai-compatible SDK
- Includes Kimi K2.6 Turbo (accounts/fireworks/routers/kimi-k2p6-turbo)
- Zero per-token cost since covered by subscription
2026-05-09 12:22:13 -05:00
EriDeLee 77eae6eef7 Update aihubmix models 2026-05-09 20:53:57 +08:00
Aiden Cline 8cbf6ed10e Merge pull request #1736 from vercel/update-vercel-models-1778258030
Update Vercel models
2026-05-08 21:58:40 -05:00
Mattia Cerutti 06d87e4411 chore(copilot): remove deprecated models 2026-05-09 00:19:50 +02:00
Frank 2cb3832618 update zen models 2026-05-08 17:11:05 -04:00
github-actions[bot] df960d1a90 chore(vercel): update Vercel model definitions
Auto-generated by weekly workflow from Vercel AI Gateway API.

Co-Authored-By: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-05-08 16:33:52 +00:00
Aiden Cline 8f2f83ef61 Merge pull request #1735 from slpdy/dev
Create DeepSeek-V4-Pro.toml
2026-05-08 10:59:33 -05:00
Aiden Cline b133426465 Merge pull request #1734 from oskarkocol/chore/update-novita-deepseek-prices
chore: update novita deepseek-v4-pro prices
2026-05-08 10:59:25 -05:00
Aiden Cline dafff5a770 Merge pull request #1730 from smakosh/add-llmgateway-models
Add new LLM Gateway text models (gpt-5.5, grok-4-3, gemini-3.1-flash-lite, qwen3.6, MiMo v2)
2026-05-08 10:58:20 -05:00
smakosh dd894f077f Merge remote-tracking branch 'upstream/dev' into add-llmgateway-models
# Conflicts:
#	providers/google/models/gemini-3.1-flash-lite.toml
2026-05-08 17:46:52 +02:00
smakosh 91590874e7 Revert "fix(models): use canonical entries for qwen3.6-max-preview and grok-4.3"
This reverts commit 70ac6fccda.
2026-05-08 17:43:07 +02:00
Jj a436236146 Create DeepSeek-V4-Pro.toml
Added DeepSeek-v4-Pro model to Nebius provider
2026-05-08 11:03:16 +01:00
oskar 1415b4be97 chore: update novita deepseek prices 2026-05-08 14:21:27 +07:00
smakosh 70ac6fccda fix(models): use canonical entries for qwen3.6-max-preview and grok-4.3
Apply the canonical TOML provided by the LLM Gateway team for the
Qwen3.6 Max Preview and Grok 4.3 parent definitions, replacing the
upstream-merged variants whose dates and pricing did not match.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-07 22:34:01 +02:00
smakosh dc3283417d Merge remote-tracking branch 'upstream/dev' into add-llmgateway-models
# Conflicts:
#	providers/alibaba/models/qwen3.6-max-preview.toml
#	providers/llmgateway/models/qwen3.6-max-preview.toml
2026-05-07 22:28:23 +02:00
smakosh 34fd6673e5 chore(llmgateway): add new text models from llmgateway catalog
Add gemini-3.1-flash-lite, grok-4-3, gpt-5.5, gpt-5.5-pro, qwen3.6
and MiMo v2 models that exist in llmgateway.io but were missing
from models.dev. Adds parent definitions for grok-4-3,
gemini-3.1-flash-lite, and qwen3.6-max-preview where they did not
already exist.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-07 22:20:52 +02:00
Isaac Huang 175082d43f Add GMI Cloud provider 2026-05-06 15:12:07 -07:00
Alex-wuhu 0f1855c0a7 provider(novita-ai): use extends for kimi k2.6 2026-05-06 10:41:46 +08:00
Shoubhit Dash 28c0d9ce23 fix(sarvam): correct output limits 2026-05-05 15:30:47 +05:30
Isaac Ng c5fbcc2c9b 📦 CHORE: remove senera 2026-05-03 16:17:37 +08:00
Isaac Ng ab2eb51b4e chore(nvidia): align Nemotron endpoint slugs
Replace stale NVIDIA Nemotron entries with the live Build catalog slugs so the local provider catalog matches current free and partner endpoints.
2026-05-03 15:47:19 +08:00
Isaac Ng 8e19ec580c 📦 CHORE: sync latest nvidia model 2026-05-03 15:19:17 +08:00
Isaac Ng 3aecc94c46 chore(nvidia): sync endpoint model catalog
Update NVIDIA model TOMLs to match the live Build endpoint list by removing stale entries and adding missing ones.

This keeps the provider catalog aligned with the current free and partner endpoint inventory.
2026-05-03 14:47:30 +08:00
Shoubhit Dash 90515f1913 provider(sarvam): add chat models 2026-05-01 14:41:47 +05:30
Alex-wuhu 6c3c4a721c provider(novita-ai): sync latest models 2026-05-01 14:19:17 +08:00
Misha Skvortsov c59c4aae6c feat(atomic-chat): add curated initial model list
Re-introduces a small curated list of models that ship preconfigured
in Atomic Chat, so opencode users get a working `models.dev` entry
out of the box instead of an empty `models: {}`.

Models (ids match the normalized form returned by Atomic Chat's
/v1/models endpoint, i.e. dots replaced with underscores):

- gemma-4-E4B-it-IQ4_XS
- gemma-4-E4B-it-MLX-4bit
- Qwen3_5-9B-Q4_K_M
- Qwen3_5-9B-MLX-4bit
- Meta-Llama-3_1-8B-Instruct-GGUF

Qwen 3.5 9B dates are taken from the verified providers/venice entry
for the same base model; quantization does not change release dates.

Made-with: Cursor
2026-04-29 17:52:56 +03:00
Mike Sukmanowsky 2cb5a99b98 fix: add model card links for Kimi K2 and Kimi K2.5 2026-04-29 09:29:36 -04:00
Mike Sukmanowsky 0c2e47e8ba Fix token limits for Amazon Bedrock Kimi K2 models
Correct context and output limits for moonshot.kimi-k2-thinking and
moonshotai.kimi-k2.5 on Amazon Bedrock:
- context: 256_000 → 262_143
- output: 256_000 → 16_000
2026-04-28 17:45:59 -04:00
Muhammad Mugni Hadi dbe92646c3 chore(chutes): add header comments to generated TOML files
Each generated TOML now includes a comment noting which fields are
auto-managed vs manually overridable on re-run.
2026-04-26 06:52:21 +07:00
Muhammad Mugni Hadi 4717c67054 feat(chutes): add API-driven model generator script
Add generate-chutes.ts that fetches models from https://llm.chutes.ai/v1/models
and generates/updates TOML files, following the same pattern as generate-vercel.ts.

Supports --dry-run, --new-only, and --keep-orphans flags. Auto-deletes TOML files
for models no longer in the API (with empty directory cleanup).

Preserves manually-set fields (family, knowledge, interleaved, status) when merging
with API data. Also syncs current models from the API.
2026-04-26 06:51:16 +07:00
Misha Skvortsov 16a8fa5c20 improve(atomic-chat): drop hardcoded model list per maintainer feedback
Made-with: Cursor
2026-04-24 17:20:26 +03:00
Misha Skvortsov 7336b3619c atomic-chat: add provider with initial blessed models
Adds Atomic Chat as a local OpenAI-compatible provider at
http://127.0.0.1:1337/v1. Includes logo and three curated models:

- unsloth/Qwen3.5-9B-IQ4_XS  (id: Qwen3_5-9B-IQ4_XS)
- unsloth/gemma-4-E4B-it-IQ4_XS  (id: gemma-4-E4B-it-IQ4_XS)
- unsloth/MiniMax-M2.5-UD-TQ1_0  (id: MiniMax-M2_5-UD-TQ1_0)

Model ids match the normalized form returned by Atomic Chat's
/v1/models endpoint (dots replaced with underscores).

Made-with: Cursor
2026-04-17 13:03:53 +03:00
1182 changed files with 13866 additions and 5294 deletions
+1
View File
@@ -35,3 +35,4 @@ jobs:
- run: bun sst deploy --stage=dev
env:
CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }}
CLOUDFLARE_DEFAULT_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_DEFAULT_ACCOUNT_ID }}
+107
View File
@@ -0,0 +1,107 @@
name: Sync Model Catalogs
on:
schedule:
- cron: "17 * * * *"
workflow_dispatch:
permissions:
contents: write
issues: write
pull-requests: write
concurrency: ${{ github.workflow }}-${{ github.ref }}
jobs:
providers:
runs-on: ubuntu-latest
outputs:
matrix: ${{ steps.providers.outputs.matrix }}
steps:
- name: Checkout code
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
with:
ref: dev
- name: Setup Bun
uses: oven-sh/setup-bun@f4d14e03ff726c06358e5557344e1da148b56cf7
with:
bun-version: latest
- name: Install dependencies
run: bun install
- name: List sync providers
id: providers
run: echo "matrix=$(bun models:sync --list-providers)" >> "$GITHUB_OUTPUT"
sync:
needs: providers
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.providers.outputs.matrix) }}
steps:
- name: Checkout code
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
with:
ref: dev
- name: Setup Bun
uses: oven-sh/setup-bun@f4d14e03ff726c06358e5557344e1da148b56cf7
with:
bun-version: latest
- name: Install dependencies
run: bun install
- name: Sync model catalogs
run: bun models:sync ${{ matrix.provider }}
env:
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
GOOGLE_GENERATIVE_AI_API_KEY: ${{ secrets.GOOGLE_GENERATIVE_AI_API_KEY }}
XAI_API_KEY: ${{ secrets.XAI_API_KEY }}
CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID }}
CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN: ${{ secrets.CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN }}
- name: Validate models
run: bun validate
- name: Create pull request
env:
GH_TOKEN: ${{ github.token }}
BRANCH: automation/sync-models-${{ matrix.provider }}
LABELS: automation,model-sync,provider:${{ matrix.provider }}
TITLE: "chore(sync): update ${{ matrix.name }} model catalog"
run: |
if [ -z "$(git status --porcelain -- providers)" ]; then
echo "No model catalog changes found."
exit 0
fi
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
git checkout -B "$BRANCH"
git add providers
git commit -m "$TITLE"
git push --force-with-lease origin "$BRANCH"
label_args=()
IFS=',' read -ra labels <<< "$LABELS"
for label in "${labels[@]}"; do
gh label create "$label" --color "0E8A16" --description "Automated model catalog sync" >/dev/null 2>&1 || true
label_args+=(--label "$label")
done
pr_number="$(gh pr list --head "$BRANCH" --base dev --json number --jq '.[0].number')"
if [ -n "$pr_number" ]; then
gh pr edit "$pr_number" --title "$TITLE" --body-file .sync/model-sync-report.md
for label in "${labels[@]}"; do
gh pr edit "$pr_number" --add-label "$label"
done
else
gh pr create --base dev --head "$BRANCH" --title "$TITLE" --body-file .sync/model-sync-report.md "${label_args[@]}"
fi
+1
View File
@@ -3,6 +3,7 @@
.idea
dist
.DS_Store
.sync/
node_modules
data/tokenspeed-monitor.sqlite
data/tokenspeed-monitor.sqlite-shm
+1
View File
File diff suppressed because one or more lines are too long
+6 -1
View File
@@ -17,11 +17,16 @@
"scripts": {
"validate": "bun ./packages/core/script/validate.ts",
"compare:migrations": "bun ./packages/core/script/compare-model-migrations.ts",
"cloudflare:sync": "bun ./packages/core/script/sync-models.ts cloudflare-workers-ai",
"chutes:generate": "bun ./packages/core/script/generate-chutes.ts",
"databricks:generate": "bun ./packages/core/script/generate-databricks.ts",
"helicone:generate": "bun ./packages/core/script/generate-helicone.ts",
"venice:generate": "bun ./packages/core/script/generate-venice.ts",
"vercel:generate": "bun ./packages/core/script/generate-vercel.ts",
"wandb:generate": "bun ./packages/core/script/generate-wandb.ts",
"digitalocean:generate": "bun ./packages/core/script/generate-digitalocean.ts"
"digitalocean:generate": "bun ./packages/core/script/generate-digitalocean.ts",
"ambient:generate": "bun ./packages/core/script/generate-ambient.ts",
"models:sync": "bun ./packages/core/script/sync-models.ts"
},
"dependencies": {
"@cloudflare/workers-types": "^4.20260424.1",
+172
View File
@@ -0,0 +1,172 @@
#!/usr/bin/env bun
/**
* Generates Ambient model TOML files from https://api.ambient.xyz/v1/models.
*
* Emits `[extends]`-format TOMLs that inherit upstream metadata
* (family, release_date, knowledge, capabilities) from the canonical
* provider model, and override only the fields Ambient's API reports:
* cost, limit, modalities.
*
* Flags:
* --dry-run Preview generated TOMLs without writing files.
*/
import { z } from "zod";
import path from "node:path";
import { mkdir } from "node:fs/promises";
const API_ENDPOINT = "https://api.ambient.xyz/v1/models";
// Allowlist for the initial rollout.
const ALLOWLIST = new Set<string>([
"zai-org/GLM-5.1-FP8",
"moonshotai/kimi-k2.6",
]);
// Maps Ambient model IDs to canonical <provider>/<model> in this repo.
// The generated TOML uses this path as `[extends].from` so capabilities
// and metadata propagate from the upstream provider automatically.
const EXTENDS_MAP: Record<string, string> = {
"zai-org/GLM-5.1-FP8": "zai/glm-5.1",
"moonshotai/kimi-k2.6": "moonshotai/kimi-k2.6",
};
const Pricing = z
.object({
prompt: z.string(),
completion: z.string(),
input_cache_read: z.string().optional(),
input_cache_write: z.string().optional(),
})
.passthrough();
const AmbientModel = z
.object({
id: z.string(),
name: z.string(),
context_length: z.number(),
max_output_length: z.number(),
input_modalities: z.array(z.string()),
output_modalities: z.array(z.string()),
pricing: Pricing,
})
.passthrough();
const AmbientResponse = z
.object({
object: z.literal("list"),
data: z.array(AmbientModel),
})
.passthrough();
const ALLOWED_MODALITIES = new Set(["text", "audio", "image", "video", "pdf"]);
function modalities(values: string[]): string[] {
return values
.map((v) => v.toLowerCase())
.filter((v) => ALLOWED_MODALITIES.has(v));
}
function perMTok(price: string): number {
const n = parseFloat(price);
if (!Number.isFinite(n)) {
throw new Error(`Invalid price: ${price}`);
}
// Round to 6 decimals to absorb float noise from per-token strings.
return Math.round(n * 1_000_000 * 1_000_000) / 1_000_000;
}
function formatToml(
model: z.infer<typeof AmbientModel>,
extendsFrom: string,
): string {
const lines: string[] = [];
lines.push("[extends]");
lines.push(`from = "${extendsFrom}"`);
lines.push("");
lines.push("[cost]");
lines.push(`input = ${perMTok(model.pricing.prompt)}`);
lines.push(`output = ${perMTok(model.pricing.completion)}`);
if (model.pricing.input_cache_read !== undefined) {
lines.push(`cache_read = ${perMTok(model.pricing.input_cache_read)}`);
}
if (model.pricing.input_cache_write !== undefined) {
lines.push(`cache_write = ${perMTok(model.pricing.input_cache_write)}`);
}
lines.push("");
lines.push("[limit]");
lines.push(`context = ${model.context_length}`);
lines.push(`output = ${model.max_output_length}`);
lines.push("");
const input = modalities(model.input_modalities);
const output = modalities(model.output_modalities);
lines.push("[modalities]");
lines.push(`input = [${input.map((m) => `"${m}"`).join(", ")}]`);
lines.push(`output = [${output.map((m) => `"${m}"`).join(", ")}]`);
return lines.join("\n") + "\n";
}
async function main() {
const dryRun = process.argv.includes("--dry-run");
const outDir = path.join(
import.meta.dirname,
"..",
"..",
"..",
"providers",
"ambient",
"models",
);
const res = await fetch(API_ENDPOINT);
if (!res.ok) {
console.error(`Fetch failed: ${res.status} ${res.statusText}`);
process.exit(1);
}
const parsed = AmbientResponse.safeParse(await res.json());
if (!parsed.success) {
console.error("Invalid Ambient response:", parsed.error.issues);
process.exit(1);
}
const selected = parsed.data.data.filter((m) => ALLOWLIST.has(m.id));
const missing = [...ALLOWLIST].filter(
(id) => !selected.some((m) => m.id === id),
);
if (missing.length > 0) {
console.error(`Allowlisted models missing from API: ${missing.join(", ")}`);
process.exit(1);
}
let count = 0;
for (const model of selected) {
const extendsFrom = EXTENDS_MAP[model.id];
if (!extendsFrom) {
console.error(`No EXTENDS_MAP entry for ${model.id}; skipping`);
continue;
}
const filePath = path.join(outDir, `${model.id}.toml`);
const toml = formatToml(model, extendsFrom);
if (dryRun) {
console.log(`--- ${path.relative(process.cwd(), filePath)} ---`);
console.log(toml);
} else {
await mkdir(path.dirname(filePath), { recursive: true });
await Bun.write(filePath, toml);
}
count++;
}
console.log(
`${dryRun ? "Previewed" : "Wrote"} ${count} model file(s) under providers/ambient/models/`,
);
}
await main();
+589
View File
@@ -0,0 +1,589 @@
#!/usr/bin/env bun
/**
* Generates Chutes model TOML files from the Chutes LLM API.
*
* Flags:
* --dry-run: Preview changes without writing files
* --new-only: Only create new models, skip updating existing ones
* --keep-orphans: Don't delete TOML files for models no longer in the API
*/
import { z } from "zod";
import path from "node:path";
import { mkdir } from "node:fs/promises";
import { ModelFamilyValues } from "../src/family.js";
const API_ENDPOINT = "https://llm.chutes.ai/v1/models";
enum SkipZeroFields {
LimitContext = "limit.context",
LimitOutput = "limit.output",
}
const Pricing = z.object({
prompt: z.number().optional(),
completion: z.number().optional(),
input_cache_read: z.number().optional(),
}).passthrough();
const ChutesModel = z.object({
id: z.string(),
created: z.number(),
pricing: Pricing.optional(),
context_length: z.number().optional(),
max_output_length: z.number().optional(),
max_model_len: z.number().optional(),
input_modalities: z.array(z.string()).optional(),
output_modalities: z.array(z.string()).optional(),
supported_features: z.array(z.string()).optional(),
supported_sampling_parameters: z.array(z.string()).optional(),
quantization: z.string().optional(),
}).passthrough();
const ChutesResponse = z.object({
data: z.array(ChutesModel),
}).passthrough();
interface ExistingModel {
name?: string;
family?: string;
attachment?: boolean;
reasoning?: boolean;
tool_call?: boolean;
structured_output?: boolean;
temperature?: boolean;
knowledge?: string;
release_date?: string;
last_updated?: string;
open_weights?: boolean;
interleaved?: boolean | { field: string };
status?: string;
cost?: {
input?: number;
output?: number;
cache_read?: number;
};
limit?: {
context?: number;
output?: number;
};
modalities?: {
input?: string[];
output?: string[];
};
}
interface MergedModel {
name: string;
family?: string;
attachment: boolean;
reasoning: boolean;
tool_call: boolean;
structured_output?: boolean;
temperature: boolean;
knowledge?: string;
release_date: string;
last_updated: string;
open_weights: boolean;
interleaved?: boolean | { field: string };
status?: string;
cost?: {
input: number;
output: number;
cache_read?: number;
};
limit: {
context: number;
output: number;
};
modalities: {
input: string[];
output: string[];
};
}
interface Changes {
field: string;
oldValue: string;
newValue: string;
}
// ── Utility functions ────────────────────────────────────────────────
function timestampToDate(timestamp: number): string {
const date = new Date(timestamp * 1000);
return date.toISOString().slice(0, 10);
}
function getTodayDate(): string {
return new Date().toISOString().slice(0, 10);
}
function formatNumber(n: number): string {
if (n >= 1000) {
return n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_");
}
return n.toString();
}
/**
* Humanize a model ID into a readable name.
* Strips the org prefix and replaces hyphens with spaces.
* e.g. "Qwen/Qwen3-32B-TEE" → "Qwen3 32B TEE"
*/
function humanizeModelName(modelId: string): string {
const parts = modelId.split("/");
const modelPart = parts[parts.length - 1];
return modelPart.replace(/-/g, " ");
}
// ── Family inference ───────────
function isSubstring(target: string, family: string): boolean {
return target.toLowerCase().includes(family.toLowerCase());
}
function matchesFamily(target: string, family: string): boolean {
const targetLower = target.toLowerCase();
const familyLower = family.toLowerCase();
let familyIdx = 0;
for (let i = 0; i < targetLower.length && familyIdx < familyLower.length; i++) {
if (targetLower[i] === familyLower[familyIdx]) {
familyIdx++;
}
}
return familyIdx === familyLower.length;
}
function inferFamily(modelId: string, modelName: string): string | undefined {
const sortedFamilies = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
// First pass: try exact substring matches
for (const family of sortedFamilies) {
if (isSubstring(modelId, family)) {
return family;
}
}
for (const family of sortedFamilies) {
if (isSubstring(modelName, family)) {
return family;
}
}
// Second pass: fall back to subsequence matching
for (const family of sortedFamilies) {
if (matchesFamily(modelId, family)) {
return family;
}
}
for (const family of sortedFamilies) {
if (matchesFamily(modelName, family)) {
return family;
}
}
return undefined;
}
// ── Load existing TOML ───────────────────────────────────────────────
async function loadExistingModel(filePath: string): Promise<ExistingModel | null> {
try {
const file = Bun.file(filePath);
if (!(await file.exists())) {
return null;
}
const toml = await import(filePath, { with: { type: "toml" } }).then(
(mod) => mod.default,
);
return toml as ExistingModel;
} catch (e) {
console.warn(`Warning: Failed to parse existing file ${filePath}:`, e);
return null;
}
}
// ── Merge API data with existing TOML ────────────────────────────────
function mergeModel(
apiModel: z.infer<typeof ChutesModel>,
existing: ExistingModel | null,
): MergedModel {
const features = new Set(apiModel.supported_features ?? []);
const samplingParams = new Set(apiModel.supported_sampling_parameters ?? []);
const inputMods = apiModel.input_modalities ?? ["text"];
const outputMods = apiModel.output_modalities ?? ["text"];
// Capabilities from API features
const hasAttachment = inputMods.some((m) =>
m === "image" || m === "video" || m === "pdf",
);
const hasReasoning = features.has("reasoning");
const hasToolCall = features.has("tools");
const hasStructuredOutput = features.has("structured_outputs");
const hasTemperature = samplingParams.size > 0
? samplingParams.has("temperature")
: true; // default true if no sampling params info
// Preserve existing values when available (manually specified)
const modelName = existing?.name ?? humanizeModelName(apiModel.id);
const family = existing?.family ?? inferFamily(apiModel.id, modelName);
const knowledge = existing?.knowledge;
const interleaved = existing?.interleaved;
const status = existing?.status;
// Release date: existing > API created timestamp > today
const releaseDate = existing?.release_date
?? timestampToDate(apiModel.created)
?? getTodayDate();
// Context limit: prefer context_length, fallback to max_model_len
const apiContext = apiModel.context_length ?? apiModel.max_model_len ?? 0;
const contextLimit = apiContext > 0
? apiContext
: (existing?.limit?.context ?? 0);
// Output limit: prefer max_output_length, fallback to existing
const apiOutput = apiModel.max_output_length ?? 0;
const outputLimit = apiOutput > 0
? apiOutput
: (existing?.limit?.output ?? 0);
const merged: MergedModel = {
name: modelName,
family,
attachment: hasAttachment,
reasoning: hasReasoning,
tool_call: hasToolCall,
temperature: hasTemperature,
release_date: releaseDate,
last_updated: getTodayDate(),
open_weights: true, // Chutes hosts open-weight models
...(hasStructuredOutput && { structured_output: hasStructuredOutput }),
...(knowledge && { knowledge }),
...(interleaved !== undefined && { interleaved }),
...(status && { status }),
limit: {
context: contextLimit,
output: outputLimit,
},
modalities: {
input: inputMods,
output: outputMods,
},
};
// Cost: API values are already in USD per 1M tokens — use directly
if (apiModel.pricing) {
const inputPrice = apiModel.pricing.prompt;
const outputPrice = apiModel.pricing.completion;
const cacheReadPrice = apiModel.pricing.input_cache_read;
if (inputPrice !== undefined && outputPrice !== undefined) {
merged.cost = {
input: inputPrice,
output: outputPrice,
...(cacheReadPrice !== undefined && { cache_read: cacheReadPrice }),
};
}
}
return merged;
}
// ── TOML formatting ──────────────────────────────────────────────────
function formatToml(model: MergedModel): string {
const lines: string[] = [];
lines.push(`# Auto-generated by generate-chutes.ts — do not edit pricing, limits, or capabilities.`);
lines.push(`# Manual overrides preserved on re-run: name, family, knowledge, interleaved, status`);
lines.push(`name = "${model.name.replace(/"/g, '\\"')}"`);
if (model.family) {
lines.push(`family = "${model.family}"`);
}
lines.push(`release_date = "${model.release_date}"`);
lines.push(`last_updated = "${model.last_updated}"`);
lines.push(`attachment = ${model.attachment}`);
lines.push(`reasoning = ${model.reasoning}`);
lines.push(`temperature = ${model.temperature}`);
lines.push(`tool_call = ${model.tool_call}`);
if (model.structured_output !== undefined) {
lines.push(`structured_output = ${model.structured_output}`);
}
lines.push(`open_weights = ${model.open_weights}`);
if (model.knowledge) {
lines.push(`knowledge = "${model.knowledge}"`);
}
if (model.status) {
lines.push(`status = "${model.status}"`);
}
if (model.cost) {
lines.push("");
lines.push(`[cost]`);
lines.push(`input = ${model.cost.input}`);
lines.push(`output = ${model.cost.output}`);
if (model.cost.cache_read !== undefined) {
lines.push(`cache_read = ${model.cost.cache_read}`);
}
}
lines.push("");
lines.push(`[limit]`);
lines.push(`context = ${formatNumber(model.limit.context)}`);
lines.push(`output = ${formatNumber(model.limit.output)}`);
lines.push("");
lines.push(`[modalities]`);
lines.push(`input = [${model.modalities.input.map((m) => `"${m}"`).join(", ")}]`);
lines.push(`output = [${model.modalities.output.map((m) => `"${m}"`).join(", ")}]`);
if (model.interleaved !== undefined) {
lines.push("");
if (model.interleaved === true) {
lines.push(`interleaved = true`);
} else if (typeof model.interleaved === "object") {
lines.push(`[interleaved]`);
lines.push(`field = "${model.interleaved.field}"`);
}
}
return lines.join("\n") + "\n";
}
// ── Change detection ─────────────────────────────────────────────────
function detectChanges(
existing: ExistingModel | null,
merged: MergedModel,
): Changes[] {
if (!existing) return [];
const changes: Changes[] = [];
const EPSILON = 0.001;
const shouldSkipZero = (field: string, oldVal: unknown, newVal: unknown): boolean => {
if (!Object.values(SkipZeroFields).includes(field as SkipZeroFields)) {
return false;
}
return (typeof oldVal === "number" && oldVal === 0) || (typeof newVal === "number" && newVal === 0);
};
const formatValue = (val: unknown): string => {
if (typeof val === "number") return formatNumber(val);
if (Array.isArray(val)) return `[${val.join(", ")}]`;
if (val === undefined) return "(none)";
return String(val);
};
const isMaterialPriceDiff = (oldPrice: unknown, newPrice: unknown): boolean => {
if (oldPrice === 0 && newPrice === undefined) return false;
if (oldPrice !== undefined && newPrice !== undefined) {
return Math.abs((oldPrice as number) - (newPrice as number)) > EPSILON;
}
return oldPrice !== newPrice;
};
const compare = (field: string, oldVal: unknown, newVal: unknown) => {
if (shouldSkipZero(field, oldVal, newVal)) return;
const isDiff = field.startsWith("cost.")
? isMaterialPriceDiff(oldVal, newVal)
: JSON.stringify(oldVal) !== JSON.stringify(newVal);
if (isDiff) {
changes.push({
field,
oldValue: formatValue(oldVal),
newValue: formatValue(newVal),
});
}
};
compare("name", existing.name, merged.name);
compare("family", existing.family, merged.family);
compare("attachment", existing.attachment, merged.attachment);
compare("reasoning", existing.reasoning, merged.reasoning);
compare("tool_call", existing.tool_call, merged.tool_call);
compare("structured_output", existing.structured_output, merged.structured_output);
compare("open_weights", existing.open_weights, merged.open_weights);
compare("release_date", existing.release_date, merged.release_date);
compare("cost.input", existing.cost?.input, merged.cost?.input);
compare("cost.output", existing.cost?.output, merged.cost?.output);
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
compare("limit.context", existing.limit?.context, merged.limit.context);
compare("limit.output", existing.limit?.output, merged.limit.output);
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
compare("modalities.output", existing.modalities?.output, merged.modalities.output);
return changes;
}
// ── Main ─────────────────────────────────────────────────────────────
async function main() {
const args = process.argv.slice(2);
const dryRun = args.includes("--dry-run");
const newOnly = args.includes("--new-only");
const keepOrphans = args.includes("--keep-orphans");
const modelsDir = path.join(
import.meta.dirname,
"..",
"..",
"..",
"providers",
"chutes",
"models",
);
console.log(`${dryRun ? "[DRY RUN] " : ""}${newOnly ? "[NEW ONLY] " : ""}${keepOrphans ? "[KEEP ORPHANS] " : ""}Fetching Chutes models from API...`);
const res = await fetch(API_ENDPOINT);
if (!res.ok) {
console.error(`Failed to fetch API: ${res.status} ${res.statusText}`);
process.exit(1);
}
const json = await res.json();
const parsed = ChutesResponse.safeParse(json);
if (!parsed.success) {
console.error("Invalid API response:", parsed.error.errors);
process.exit(1);
}
const apiModels = parsed.data.data;
// Scan existing TOML files
const existingFiles = new Set<string>();
try {
for await (const file of new Bun.Glob("**/*.toml").scan({
cwd: modelsDir,
absolute: false,
})) {
existingFiles.add(file);
}
} catch {
}
console.log(`Found ${apiModels.length} models in API, ${existingFiles.size} existing files\n`);
const apiModelIds = new Set<string>();
let created = 0;
let updated = 0;
let unchanged = 0;
for (const apiModel of apiModels) {
const relativePath = `${apiModel.id}.toml`;
const filePath = path.join(modelsDir, relativePath);
const dirPath = path.dirname(filePath);
apiModelIds.add(relativePath);
const existing = await loadExistingModel(filePath);
const merged = mergeModel(apiModel, existing);
const tomlContent = formatToml(merged);
if (existing === null) {
created++;
if (dryRun) {
console.log(`[DRY RUN] Would create: ${relativePath}`);
console.log(` name = "${merged.name}"`);
if (merged.family) {
console.log(` family = "${merged.family}" (inferred)`);
}
console.log("");
} else {
await mkdir(dirPath, { recursive: true });
await Bun.write(filePath, tomlContent);
console.log(`Created: ${relativePath}`);
}
} else {
if (newOnly) {
unchanged++;
continue;
}
const changes = detectChanges(existing, merged);
const existingContent = await Bun.file(filePath).text();
const formatChanged = existingContent !== tomlContent;
if (changes.length > 0 || formatChanged) {
updated++;
if (dryRun) {
console.log(`[DRY RUN] Would update: ${relativePath}`);
} else {
await mkdir(dirPath, { recursive: true });
await Bun.write(filePath, tomlContent);
console.log(`Updated: ${relativePath}`);
}
for (const change of changes) {
console.log(` ${change.field}: ${change.oldValue}${change.newValue}`);
}
if (changes.length === 0 && formatChanged) {
console.log(` (format-only change)`);
}
console.log("");
} else {
unchanged++;
}
}
}
// Handle orphaned files (on disk but not in API)
const orphaned: string[] = [];
for (const file of existingFiles) {
if (!apiModelIds.has(file)) {
orphaned.push(file);
const orphanPath = path.join(modelsDir, file);
if (keepOrphans) {
console.log(`Orphaned (kept): ${file}`);
} else if (dryRun) {
console.log(`[DRY RUN] Would delete: ${file}`);
} else {
await Bun.file(orphanPath).delete();
console.log(`Deleted: ${file}`);
// Clean up empty parent directories
const parentDir = path.dirname(orphanPath);
try {
const remaining = [];
for await (const entry of new Bun.Glob("*").scan({ cwd: parentDir })) {
remaining.push(entry);
}
if (remaining.length === 0) {
const { rmdir } = await import("node:fs/promises");
await rmdir(parentDir);
console.log(` Removed empty directory: ${path.basename(parentDir)}/`);
}
} catch {
// Directory not empty or other error, ignore
}
}
}
}
console.log("");
if (dryRun) {
console.log(
`Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} would be deleted`,
);
} else if (keepOrphans) {
console.log(
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned (kept)`,
);
} else {
console.log(
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} deleted`,
);
}
}
await main();
+287
View File
@@ -0,0 +1,287 @@
#!/usr/bin/env bun
/**
* Generates Databricks model TOML files from the Foundation Model API endpoint.
*
* Each Databricks endpoint exposes a model from another provider (Anthropic,
* OpenAI, Google, etc.), so the generated TOML uses [extends] to inherit
* canonical metadata from that upstream provider's TOML in models.dev.
*
* Usage:
* DATABRICKS_HOST=<host> DATABRICKS_TOKEN=<pat> bun run databricks:generate
* bun run databricks:generate --workspace <host> --token <pat>
*
* Flags:
* --dry-run: Preview changes without writing files
* --new-only: Only create new models, skip updating existing ones
*/
import { z } from "zod";
import path from "node:path";
import { mkdir, readFile } from "node:fs/promises";
import { existsSync } from "node:fs";
const args = process.argv.slice(2);
const flag = (name: string) => {
const i = args.indexOf(`--${name}`);
return i !== -1 ? args[i + 1] : undefined;
};
const dryRun = args.includes("--dry-run");
const newOnly = args.includes("--new-only");
const host = flag("workspace") ?? process.env.DATABRICKS_HOST;
const token = flag("token") ?? process.env.DATABRICKS_TOKEN;
if (!host || !token) {
console.error(
"Usage: DATABRICKS_HOST=<host> DATABRICKS_TOKEN=<pat> bun run databricks:generate",
);
process.exit(1);
}
const workspace = host.replace(/^https?:\/\//, "").replace(/\/$/, "");
const PROVIDERS_DIR = path.join(import.meta.dirname, "..", "..", "..", "providers");
const MODELS_DIR = path.join(PROVIDERS_DIR, "databricks", "models");
// ---------------------------------------------------------------------------
// API schemas
// ---------------------------------------------------------------------------
const FoundationModel = z
.object({
ai_gateway_v2_supported: z.boolean().optional(),
api_types: z.array(z.string()).optional(),
})
.passthrough();
const ServedEntity = z
.object({
foundation_model: FoundationModel.optional(),
})
.passthrough();
const Endpoint = z
.object({
name: z.string(),
config: z
.object({
served_entities: z.array(ServedEntity).optional(),
})
.passthrough()
.optional(),
})
.passthrough();
const FoundationModelsResponse = z
.object({
endpoints: z.array(Endpoint),
})
.passthrough();
// ---------------------------------------------------------------------------
// Canonical resolution: map a Databricks endpoint name to a models.dev entry
// ---------------------------------------------------------------------------
const PREFIX_TO_PROVIDER: [string, string][] = [
["claude-", "anthropic"],
["gpt-", "openai"],
["gemini-", "google"],
["mistral-", "mistral"],
["mixtral-", "mistral"],
];
type Resolution =
| { type: "extends"; from: string }
| { type: "inline"; content: string }
| null;
async function resolveCanonical(endpointName: string): Promise<Resolution> {
const bare = endpointName.replace(/^databricks-/, "");
// Models in provider subdirectories (e.g. openrouter/openai/gpt-oss-*)
// can't use [extends] (schema requires provider/model format), so inline.
if (bare.startsWith("gpt-oss-")) {
const p = path.join(PROVIDERS_DIR, "openrouter", "models", "openai", `${bare}.toml`);
if (existsSync(p)) {
return { type: "inline", content: await readFile(p, "utf8") };
}
}
// Meta Llama: "meta-llama-3-3-70b-instruct" → "llama-3.3-70b-instruct"
if (bare.startsWith("meta-llama-") || bare.startsWith("llama-")) {
const llamaId = bare
.replace(/^meta-llama-/, "llama-")
.replace(/^(llama-\d+)-(\d+)-/, "$1.$2-");
const p = path.join(PROVIDERS_DIR, "llama", "models", `${llamaId}.toml`);
if (existsSync(p)) return { type: "extends", from: `llama/${llamaId}` };
}
for (const [prefix, provider] of PREFIX_TO_PROVIDER) {
if (!bare.startsWith(prefix)) continue;
const exact = path.join(PROVIDERS_DIR, provider, "models", `${bare}.toml`);
if (existsSync(exact)) return { type: "extends", from: `${provider}/${bare}` };
// Try with hyphens-as-dots in version (e.g. gpt-5-4 → gpt-5.4)
const dotted = bare.replace(/^((?:[a-z]+-)+\d+)-(\d)/, "$1.$2");
if (dotted !== bare) {
const dottedExact = path.join(PROVIDERS_DIR, provider, "models", `${dotted}.toml`);
if (existsSync(dottedExact)) return { type: "extends", from: `${provider}/${dotted}` };
}
// Fuzzy: longest filename that shares a prefix with bare or its dotted form
const candidates = [bare, ...(dotted !== bare ? [dotted] : [])];
const files: string[] = [];
try {
for await (const f of new Bun.Glob("*.toml").scan({
cwd: path.join(PROVIDERS_DIR, provider, "models"),
})) {
files.push(f);
}
} catch {
// provider directory may not exist
}
const match = files
.map((f) => f.replace(/\.toml$/, ""))
.filter((id) => candidates.some((c) => id.startsWith(c) || c.startsWith(id)))
.sort((a, b) => b.length - a.length)[0];
if (match) return { type: "extends", from: `${provider}/${match}` };
}
return null;
}
function formatToml(resolution: Resolution, endpointName: string): string {
if (resolution?.type === "extends") {
return `[extends]\nfrom = "${resolution.from}"\n`;
}
if (resolution?.type === "inline") {
return resolution.content;
}
return `# TODO: fill in details for ${endpointName}\nname = "${endpointName}"\n`;
}
// ---------------------------------------------------------------------------
// Main
// ---------------------------------------------------------------------------
const IGNORE_PREFIXES = [
"databricks-llama-",
"databricks-meta-llama-",
"databricks-qwen",
"databricks-gemma-",
];
async function main() {
console.log(
`${dryRun ? "[DRY RUN] " : ""}${newOnly ? "[NEW ONLY] " : ""}Fetching Databricks foundation-models...`,
);
const url = `https://${workspace}/api/2.0/serving-endpoints:foundation-models`;
const res = await fetch(url, {
headers: { Authorization: `Bearer ${token}` },
});
if (!res.ok) {
console.error(`Failed to fetch API: ${res.status} ${res.statusText}`);
console.error(await res.text().catch(() => ""));
process.exit(1);
}
const json = await res.json();
const parsed = FoundationModelsResponse.safeParse(json);
if (!parsed.success) {
console.error("Invalid API response:", parsed.error.errors);
process.exit(1);
}
const endpoints = parsed.data.endpoints.filter(
(e) =>
!IGNORE_PREFIXES.some((p) => e.name.startsWith(p)) &&
e.config?.served_entities?.some(
(se) =>
se.foundation_model?.ai_gateway_v2_supported === true &&
se.foundation_model?.api_types?.includes("mlflow/v1/chat/completions"),
),
);
const existingFiles = new Set<string>();
try {
for await (const f of new Bun.Glob("*.toml").scan({ cwd: MODELS_DIR })) {
existingFiles.add(f);
}
} catch {
// directory may not exist yet
}
console.log(
`Found ${endpoints.length} models in API, ${existingFiles.size} existing files\n`,
);
const apiModelIds = new Set<string>();
let created = 0;
let updated = 0;
let unchanged = 0;
for (const ep of endpoints) {
const filename = `${ep.name}.toml`;
apiModelIds.add(filename);
const filePath = path.join(MODELS_DIR, filename);
const resolution = await resolveCanonical(ep.name);
const newContent = formatToml(resolution, ep.name);
const tag = resolution?.type === "extends" ? `extends ${resolution.from}` : resolution?.type ?? "stub";
const existed = existsSync(filePath);
if (!existed) {
created++;
if (dryRun) {
console.log(`[DRY RUN] Would create: ${filename}${tag}`);
} else {
await mkdir(MODELS_DIR, { recursive: true });
await Bun.write(filePath, newContent);
console.log(`Created: ${filename}${tag}`);
}
continue;
}
if (newOnly) {
unchanged++;
continue;
}
const existingContent = await readFile(filePath, "utf8");
if (existingContent === newContent) {
unchanged++;
continue;
}
updated++;
if (dryRun) {
console.log(`[DRY RUN] Would update: ${filename}${tag}`);
} else {
await Bun.write(filePath, newContent);
console.log(`Updated: ${filename}${tag}`);
}
}
const orphaned: string[] = [];
for (const file of existingFiles) {
if (!apiModelIds.has(file)) {
orphaned.push(file);
console.log(`Warning: Orphaned file (not in API): ${file}`);
}
}
console.log("");
if (dryRun) {
console.log(
`Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} orphaned`,
);
} else {
console.log(
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned`,
);
}
}
await main();
+61 -15
View File
@@ -209,7 +209,18 @@ interface ExistingModel {
output?: number;
cache_read?: number;
cache_write?: number;
context_min?: number;
};
tiers?: Array<{
tier: {
type?: "context";
size: number;
};
input?: number;
output?: number;
cache_read?: number;
cache_write?: number;
}>;
};
limit?: {
context?: number;
@@ -262,6 +273,7 @@ interface MergedModel {
output: number;
cache_read?: number;
cache_write?: number;
context_min?: number;
};
};
limit: {
@@ -310,6 +322,35 @@ function inferFamily(modelId: string, modelName: string): string | undefined {
return undefined;
}
function getExistingLongContextCost(existing: ExistingModel | null) {
const tier = existing?.cost?.tiers?.find(
(tier) =>
(tier.tier.type === undefined || tier.tier.type === "context") &&
tier.tier.size >= 200_000,
);
if (tier) {
return {
...tier,
context_min: tier.tier.size,
};
}
return existing?.cost?.context_over_200k === undefined
? undefined
: {
...existing.cost.context_over_200k,
context_min: 200_000,
};
}
function getLongContextMin(cost: { context_min?: number }) {
return cost.context_min ?? 200_000;
}
function formatInlineNumber(n: number): string {
return n >= 1000 ? n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_") : n.toString();
}
// ---------------------------------------------------------------------------
// Merge API data with existing TOML
// ---------------------------------------------------------------------------
@@ -394,27 +435,30 @@ function mergeModel(
};
// Context-tiered pricing (>200k) from the static-content API
const existingLongContextCost = getExistingLongContextCost(existing);
if (pricing?.inputOver200k !== undefined && pricing?.outputOver200k !== undefined) {
merged.cost.context_over_200k = {
input: pricing.inputOver200k,
output: pricing.outputOver200k,
...(existing?.cost?.context_over_200k?.cache_read !== undefined && {
cache_read: existing.cost.context_over_200k.cache_read,
context_min: existingLongContextCost?.context_min ?? 200_000,
...(existingLongContextCost?.cache_read !== undefined && {
cache_read: existingLongContextCost.cache_read,
}),
...(existing?.cost?.context_over_200k?.cache_write !== undefined && {
cache_write: existing.cost.context_over_200k.cache_write,
...(existingLongContextCost?.cache_write !== undefined && {
cache_write: existingLongContextCost.cache_write,
}),
};
} else if (existing?.cost?.context_over_200k) {
// Preserve manually-entered context_over_200k if API has no data
} else if (existingLongContextCost) {
// Preserve manually-entered tiered pricing if API has no data
merged.cost.context_over_200k = {
input: existing.cost.context_over_200k.input ?? inputPrice,
output: existing.cost.context_over_200k.output ?? outputPrice,
...(existing.cost.context_over_200k.cache_read !== undefined && {
cache_read: existing.cost.context_over_200k.cache_read,
input: existingLongContextCost.input ?? inputPrice,
output: existingLongContextCost.output ?? outputPrice,
context_min: existingLongContextCost.context_min,
...(existingLongContextCost.cache_read !== undefined && {
cache_read: existingLongContextCost.cache_read,
}),
...(existing.cost.context_over_200k.cache_write !== undefined && {
cache_write: existing.cost.context_over_200k.cache_write,
...(existingLongContextCost.cache_write !== undefined && {
cache_write: existingLongContextCost.cache_write,
}),
};
}
@@ -463,7 +507,8 @@ function formatToml(model: MergedModel): string {
if (model.cost.context_over_200k) {
lines.push("");
lines.push(`[cost.context_over_200k]`);
lines.push(`[[cost.tiers]]`);
lines.push(`tier = { size = ${formatInlineNumber(getLongContextMin(model.cost.context_over_200k))} }`);
lines.push(`input = ${model.cost.context_over_200k.input}`);
lines.push(`output = ${model.cost.context_over_200k.output}`);
if (model.cost.context_over_200k.cache_read !== undefined)
@@ -524,8 +569,9 @@ function detectChanges(existing: ExistingModel | null, merged: MergedModel): Cha
compare("status", existing.status, merged.status);
compare("cost.input", existing.cost?.input, merged.cost?.input);
compare("cost.output", existing.cost?.output, merged.cost?.output);
compare("cost.context_over_200k.input", existing.cost?.context_over_200k?.input, merged.cost?.context_over_200k?.input);
compare("cost.context_over_200k.output", existing.cost?.context_over_200k?.output, merged.cost?.context_over_200k?.output);
const existingLongContextCost = getExistingLongContextCost(existing);
compare("cost.context_over_200k.input", existingLongContextCost?.input, merged.cost?.context_over_200k?.input);
compare("cost.context_over_200k.output", existingLongContextCost?.output, merged.cost?.context_over_200k?.output);
compare("limit.context", existing.limit?.context, merged.limit.context);
compare("limit.output", existing.limit?.output, merged.limit.output);
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
+44 -5
View File
@@ -162,7 +162,18 @@ interface ExistingModel {
output?: number;
cache_read?: number;
cache_write?: number;
context_min?: number;
};
tiers?: Array<{
tier: {
type?: "context";
size: number;
};
input?: number;
output?: number;
cache_read?: number;
cache_write?: number;
}>;
};
limit?: {
context?: number;
@@ -195,6 +206,30 @@ async function loadExistingModel(filePath: string): Promise<ExistingModel | null
}
}
function getExistingLongContextMin(existing: ExistingModel | null) {
return (
existing?.cost?.tiers?.find(
(tier) =>
(tier.tier.type === undefined || tier.tier.type === "context") &&
tier.tier.size >= 200_000,
)?.tier.size ?? 200_000
);
}
function getExistingLongContextCost(existing: ExistingModel | null) {
return (
existing?.cost?.tiers?.find(
(tier) =>
(tier.tier.type === undefined || tier.tier.type === "context") &&
tier.tier.size >= 200_000,
) ?? existing?.cost?.context_over_200k
);
}
function getLongContextMin(cost: { context_min?: number }) {
return cost.context_min ?? 200_000;
}
interface MergedModel {
name: string;
family?: string;
@@ -219,6 +254,7 @@ interface MergedModel {
output: number;
cache_read?: number;
cache_write?: number;
context_min?: number;
};
};
limit: {
@@ -292,6 +328,7 @@ function mergeModel(
merged.cost.context_over_200k = {
input: spec.pricing.extended.input.usd,
output: spec.pricing.extended.output.usd,
context_min: spec.pricing.extended.context_token_threshold,
...(spec.pricing.extended.cache_input && { cache_read: spec.pricing.extended.cache_input.usd }),
...(spec.pricing.extended.cache_write && { cache_write: spec.pricing.extended.cache_write.usd }),
};
@@ -366,7 +403,8 @@ function formatToml(model: MergedModel): string {
if (model.cost.context_over_200k) {
lines.push("");
lines.push(`[cost.context_over_200k]`);
lines.push(`[[cost.tiers]]`);
lines.push(`tier = { size = ${formatNumber(getLongContextMin(model.cost.context_over_200k))} }`);
lines.push(`input = ${model.cost.context_over_200k.input}`);
lines.push(`output = ${model.cost.context_over_200k.output}`);
if (model.cost.context_over_200k.cache_read !== undefined) {
@@ -438,10 +476,11 @@ function detectChanges(
compare("cost.output", existing.cost?.output, merged.cost?.output);
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
compare("cost.cache_write", existing.cost?.cache_write, merged.cost?.cache_write);
compare("cost.context_over_200k.input", existing.cost?.context_over_200k?.input, merged.cost?.context_over_200k?.input);
compare("cost.context_over_200k.output", existing.cost?.context_over_200k?.output, merged.cost?.context_over_200k?.output);
compare("cost.context_over_200k.cache_read", existing.cost?.context_over_200k?.cache_read, merged.cost?.context_over_200k?.cache_read);
compare("cost.context_over_200k.cache_write", existing.cost?.context_over_200k?.cache_write, merged.cost?.context_over_200k?.cache_write);
const existingLongContextCost = getExistingLongContextCost(existing);
compare("cost.context_over_200k.input", existingLongContextCost?.input, merged.cost?.context_over_200k?.input);
compare("cost.context_over_200k.output", existingLongContextCost?.output, merged.cost?.context_over_200k?.output);
compare("cost.context_over_200k.cache_read", existingLongContextCost?.cache_read, merged.cost?.context_over_200k?.cache_read);
compare("cost.context_over_200k.cache_write", existingLongContextCost?.cache_write, merged.cost?.context_over_200k?.cache_write);
compare("limit.context", existing.limit?.context, merged.limit.context);
compare("limit.output", existing.limit?.output, merged.limit.output);
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
+5
View File
@@ -0,0 +1,5 @@
#!/usr/bin/env bun
import { main } from "../src/sync/index.js";
await main();
+4
View File
@@ -55,6 +55,7 @@ export const ModelFamilyValues = [
"deepseek",
"deepseek-thinking",
"deepseek-flash",
"deepseek-flash-free",
"deepseek-flash-think",
// Microsoft Phi
@@ -81,6 +82,7 @@ export const ModelFamilyValues = [
// xAI Grok
"grok",
"grok-build",
"grok-vision",
"grok-beta",
@@ -289,12 +291,14 @@ export const ModelFamilyValues = [
"rnj",
// Tecent Hy
"hy3",
"hy3-free",
// Ling & Ring (InclusionAI)
"ling",
"ling-flash-free",
"ring",
"ring-1t-free",
// Kat Coder
"kat-coder",
+53 -5
View File
@@ -2,9 +2,9 @@ import path from "path";
import { mergeDeep } from "remeda";
import { z } from "zod";
import { Provider, Model } from "./schema.js";
import { Provider, Model, AuthoredModel, AuthoredModelShape } from "./schema.js";
const ExtendsModel = Model.sourceType()
const ExtendsModel = AuthoredModelShape
.partial()
.extend({
extends: z
@@ -71,12 +71,12 @@ export async function generate(directory: string) {
});
continue;
}
const model = Model.safeParse(toml);
const model = AuthoredModel.safeParse(toml);
if (!model.success) {
model.error.cause = { modelPath, toml };
throw model.error;
}
provider.data.models[modelID] = model.data;
provider.data.models[modelID] = normalizeModelCost(model.data);
}
result[providerID] = provider.data;
}
@@ -144,7 +144,7 @@ export async function generate(directory: string) {
}
}
const model = Model.safeParse(merged);
const model = Model.safeParse(normalizeCost(merged));
if (!model.success) {
model.error.cause = { modelPath: pendingModel.modelPath, toml: merged };
throw model.error;
@@ -155,3 +155,51 @@ export async function generate(directory: string) {
return result;
}
function normalizeModelCost(model: z.infer<typeof AuthoredModel>): Model {
return normalizeCost(model) as Model;
}
function normalizeCost(model: Record<string, unknown>) {
const cost = model.cost;
if (cost === undefined || cost === null || typeof cost !== "object" || Array.isArray(cost)) {
return model;
}
const tiers = (cost as { tiers?: unknown }).tiers;
if (!Array.isArray(tiers)) {
return model;
}
if (tiers.length !== 1) {
return model;
}
const contextOver200k = tiers.find((tier) => {
if (tier === null || typeof tier !== "object" || Array.isArray(tier)) return false;
const tierConfig = (tier as { tier?: unknown }).tier;
if (tierConfig === null || typeof tierConfig !== "object" || Array.isArray(tierConfig)) return false;
const type = (tierConfig as { type?: unknown }).type;
const size = (tierConfig as { size?: unknown }).size;
// context_over_200k is a legacy compatibility field. It intentionally
// includes higher thresholds; cost.tiers carries the exact threshold.
return (
(type === undefined || type === "context") &&
typeof size === "number" &&
size >= 200_000
);
});
if (contextOver200k === undefined) {
return model;
}
const { tier: _tier, ...legacyCost } = contextOver200k as Record<string, unknown>;
return {
...model,
cost: {
...(cost as Record<string, unknown>),
context_over_200k: legacyCost,
},
};
}
+151 -97
View File
@@ -21,110 +21,164 @@ const JsonValue: z.ZodType<JsonValue> = z.lazy(() =>
]),
);
const Cost = z.object({
input: z.number().min(0, "Input price cannot be negative"),
output: z.number().min(0, "Output price cannot be negative"),
reasoning: z.number().min(0, "Input price cannot be negative").optional(),
cache_read: z
.number()
.min(0, "Cache read price cannot be negative")
const Cost = z
.object({
input: z.number().min(0, "Input price cannot be negative"),
output: z.number().min(0, "Output price cannot be negative"),
reasoning: z
.number()
.min(0, "Reasoning price cannot be negative")
.optional(),
cache_read: z
.number()
.min(0, "Cache read price cannot be negative")
.optional(),
cache_write: z
.number()
.min(0, "Cache write price cannot be negative")
.optional(),
input_audio: z
.number()
.min(0, "Audio input price cannot be negative")
.optional(),
output_audio: z
.number()
.min(0, "Audio output price cannot be negative")
.optional(),
});
const CostTier = Cost.extend({
tier: z
.object({
type: z.literal("context").default("context"),
size: z.number().int().min(0, "Context tier size cannot be negative"),
})
.strict(),
}).strict();
const AuthoredCost = Cost.extend({
context_over_200k: z.never().optional(),
tiers: z.array(CostTier).optional(),
});
const OutputCost = Cost.extend({
context_over_200k: Cost.optional(),
tiers: z.array(CostTier).optional(),
});
const ModelBase = z.object({
id: z.string(),
name: z.string().min(1, "Model name cannot be empty"),
family: ModelFamily.optional(),
attachment: z.boolean(),
reasoning: z.boolean(),
tool_call: z.boolean(),
interleaved: z
.union([
z.literal(true),
z
.object({
field: z.enum(["reasoning_content", "reasoning_details"]),
})
.strict(),
])
.optional(),
cache_write: z
.number()
.min(0, "Cache write price cannot be negative")
structured_output: z.boolean().optional(),
temperature: z.boolean().optional(),
knowledge: z
.string()
.regex(/^\d{4}-\d{2}(-\d{2})?$/, {
message: "Must be in YYYY-MM or YYYY-MM-DD format",
})
.optional(),
input_audio: z
.number()
.min(0, "Audio input price cannot be negative")
release_date: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
message: "Must be in YYYY-MM or YYYY-MM-DD format",
}),
last_updated: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
message: "Must be in YYYY-MM or YYYY-MM-DD format",
}),
modalities: z.object({
input: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
output: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
}),
open_weights: z.boolean(),
limit: z.object({
context: z.number().min(0, "Context window must be positive"),
input: z.number().min(0, "Input tokens must be positive").optional(),
output: z.number().min(0, "Output tokens must be positive"),
}),
status: z.enum(["alpha", "beta", "deprecated"]).optional(),
experimental: z
.object({
modes: z
.record(
z.object({
cost: Cost.optional(),
provider: z
.object({
body: z.record(JsonValue).optional(),
headers: z.record(z.string()).optional(),
})
.optional(),
}),
)
.optional(),
})
.optional(),
output_audio: z
.number()
.min(0, "Audio output price cannot be negative")
provider: z
.object({
npm: z.string().optional(),
api: z.string().optional(),
shape: z.enum(["responses", "completions"]).optional(),
body: z.record(JsonValue).optional(),
headers: z.record(z.string()).optional(),
})
.optional(),
});
export const Model = z
function refineModel<T extends z.ZodTypeAny>(schema: T) {
return schema
.refine(
(data) => {
return !(data.reasoning === false && data.cost?.reasoning !== undefined);
},
{
message: "Cannot set cost.reasoning when reasoning is false",
path: ["cost", "reasoning"],
},
)
.refine(
(data) => {
const tiers = data.cost?.tiers;
if (tiers === undefined) return true;
const sizes = tiers.map((tier: { tier: { size: number } }) => tier.tier.size);
return new Set(sizes).size === sizes.length;
},
{
message: "Cost context tiers must not have duplicate sizes",
path: ["cost", "tiers"],
},
);
}
export const ModelShape = z
.object({
id: z.string(),
name: z.string().min(1, "Model name cannot be empty"),
family: ModelFamily.optional(),
attachment: z.boolean(),
reasoning: z.boolean(),
tool_call: z.boolean(),
interleaved: z
.union([
z.literal(true),
z
.object({
field: z.enum(["reasoning_content", "reasoning_details"]),
})
.strict(),
])
.optional(),
structured_output: z.boolean().optional(),
temperature: z.boolean().optional(),
knowledge: z
.string()
.regex(/^\d{4}-\d{2}(-\d{2})?$/, {
message: "Must be in YYYY-MM or YYYY-MM-DD format",
})
.optional(),
release_date: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
message: "Must be in YYYY-MM or YYYY-MM-DD format",
}),
last_updated: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
message: "Must be in YYYY-MM or YYYY-MM-DD format",
}),
modalities: z.object({
input: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
output: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
}),
open_weights: z.boolean(),
cost: Cost.extend({
context_over_200k: Cost.optional(),
}).optional(),
limit: z.object({
context: z.number().min(0, "Context window must be positive"),
input: z.number().min(0, "Input tokens must be positive").optional(),
output: z.number().min(0, "Output tokens must be positive"),
}),
status: z.enum(["alpha", "beta", "deprecated"]).optional(),
experimental: z
.object({
modes: z
.record(
z.object({
cost: Cost.optional(),
provider: z
.object({
body: z.record(JsonValue).optional(),
headers: z.record(z.string()).optional(),
})
.optional(),
}),
)
.optional(),
})
.optional(),
provider: z
.object({
npm: z.string().optional(),
api: z.string().optional(),
shape: z.enum(["responses", "completions"]).optional(),
body: z.record(JsonValue).optional(),
headers: z.record(z.string()).optional(),
})
.optional(),
...ModelBase.shape,
cost: OutputCost.optional(),
})
.strict()
.refine(
(data) => {
return !(data.reasoning === false && data.cost?.reasoning !== undefined);
},
{
message: "Cannot set cost.reasoning when reasoning is false",
path: ["cost", "reasoning"],
},
);
.strict();
export const AuthoredModelShape = z
.object({
...ModelBase.shape,
cost: AuthoredCost.optional(),
})
.strict();
export const Model = refineModel(ModelShape);
export const AuthoredModel = refineModel(AuthoredModelShape);
export type Model = z.infer<typeof Model>;
+481
View File
@@ -0,0 +1,481 @@
import path from "node:path";
import { mkdir, readdir, rm } from "node:fs/promises";
import { z } from "zod";
import { AuthoredModel, AuthoredModelShape } from "../schema.js";
import { cloudflareWorkersAi } from "./providers/cloudflare-workers-ai.js";
import { google } from "./providers/google.js";
import { openrouter } from "./providers/openrouter.js";
import { xai } from "./providers/xai.js";
const ExtendsConfig = z
.object({
from: z.string(),
omit: z.array(z.string()).optional(),
})
.strict();
const ExistingExtendsConfig = z
.object({
from: z.string(),
omit: z.array(z.string()).optional(),
})
.passthrough();
const ExistingModel = AuthoredModelShape.partial()
.extend({
extends: ExistingExtendsConfig.optional(),
})
.strict();
const SyncedExtendsModel = AuthoredModelShape.partial()
.extend({
id: z.string(),
extends: ExtendsConfig,
})
.strict();
const SyncedAuthoredModel = z.union([AuthoredModel, SyncedExtendsModel]);
export type ExistingModel = z.infer<typeof ExistingModel>;
export type SyncedFullModel = Omit<z.infer<typeof AuthoredModelShape>, "id">;
export type SyncedExtendsModel = Omit<z.infer<typeof SyncedExtendsModel>, "id">;
export type SyncedModel = SyncedFullModel | SyncedExtendsModel;
export interface SyncProvider<SourceModel> {
id: string;
name: string;
modelsDir: string;
skipCreates?: boolean;
sourceID?(model: SourceModel): string;
skippedNotice?(ids: string[]): string[];
fetchModels(): Promise<unknown>;
parseModels(raw: unknown): SourceModel[];
translateModel(
model: SourceModel,
context: { existing(id: string): ExistingModel | undefined },
): { id: string; model: SyncedModel } | undefined;
}
export interface SyncResult {
id: string;
name: string;
status: "changed" | "unchanged";
created: number;
updated: number;
deleted: number;
unchanged: number;
notices: string[];
files: Array<{ status: "created" | "updated" | "deleted"; path: string }>;
}
export const providers: {
"cloudflare-workers-ai": SyncProvider<any>;
google: SyncProvider<any>;
openrouter: SyncProvider<any>;
xai: SyncProvider<any>;
} = {
"cloudflare-workers-ai": cloudflareWorkersAi,
google,
openrouter,
xai,
};
export const groups = {
aggregators: ["openrouter"],
cloudflare: ["cloudflare-workers-ai"],
direct: ["google", "xai"],
} as const;
type ProviderID = keyof typeof providers;
interface SyncOptions {
dryRun?: boolean;
newOnly?: boolean;
}
export async function syncProviderByID(id: ProviderID, options: SyncOptions = {}) {
return syncProvider(providers[id], options);
}
export async function syncProvider<SourceModel>(
provider: SyncProvider<SourceModel>,
options: SyncOptions = {},
): Promise<SyncResult> {
console.log(`\nSyncing ${provider.name}...`);
const existing = await readExisting(provider.modelsDir);
const sourceModels = provider.parseModels(await provider.fetchModels());
const desired = new Map<string, { model: z.infer<typeof SyncedAuthoredModel>; content: string }>();
const skippedRemote: string[] = [];
for (const sourceModel of sourceModels) {
const translated = provider.translateModel(sourceModel, {
existing(id) {
return existing.get(`${id}.toml`)?.toml;
},
});
if (translated === undefined) {
if (provider.skipCreates) skippedRemote.push(provider.sourceID?.(sourceModel) ?? "unknown");
continue;
}
const relativePath = `${translated.id}.toml`;
if (provider.skipCreates && !existing.has(relativePath)) {
skippedRemote.push(translated.id);
continue;
}
if (desired.has(relativePath)) {
throw new Error(`Duplicate synced model path: ${provider.id}/${relativePath}`);
}
const parsed = SyncedAuthoredModel.safeParse({
id: translated.id,
...translated.model,
});
if (!parsed.success) {
parsed.error.cause = { provider: provider.id, path: relativePath };
throw parsed.error;
}
desired.set(relativePath, {
model: parsed.data,
content: formatToml(parsed.data),
});
}
const files: SyncResult["files"] = [];
let unchanged = 0;
for (const [relativePath, file] of desired) {
const filePath = path.join(provider.modelsDir, relativePath);
const current = existing.get(relativePath);
if (current === undefined) {
files.push({ status: "created", path: filePath });
if (options.dryRun) {
console.log(`Would create ${relativePath}`);
} else {
await mkdir(path.dirname(filePath), { recursive: true });
await Bun.write(filePath, file.content);
}
continue;
}
if (!sameModel(relativePath, current.toml, file.model)) {
if (options.newOnly) {
unchanged++;
continue;
}
files.push({ status: "updated", path: filePath });
if (options.dryRun) {
console.log(`Would update ${relativePath}`);
} else {
if (current.symlink) await rm(filePath, { force: true });
await Bun.write(filePath, file.content);
}
} else {
unchanged++;
}
}
for (const relativePath of existing.keys()) {
if (desired.has(relativePath)) continue;
if (options.newOnly) {
console.log(`Skipping removal in new-only mode: ${relativePath}`);
unchanged++;
continue;
}
const filePath = path.join(provider.modelsDir, relativePath);
files.push({ status: "deleted", path: filePath });
if (options.dryRun) {
console.log(`Would remove ${relativePath}`);
} else {
await rm(filePath, { force: true });
}
}
const result = summarize(provider, files, unchanged, provider.skippedNotice?.(skippedRemote) ?? []);
console.log(
`${options.dryRun ? "Dry run: " : ""}${result.created} created, ${result.updated} updated, ${result.deleted} removed, ${result.unchanged} unchanged`,
);
return result;
}
export async function syncTargets(target: string, options: SyncOptions = {}) {
const ids = target in groups
? groups[target as keyof typeof groups]
: target in providers
? [target as ProviderID]
: undefined;
if (ids === undefined) {
throw new Error(`Unknown sync target: ${target}`);
}
const results: SyncResult[] = [];
for (const id of ids) {
results.push(await syncProviderByID(id as ProviderID, options));
}
return results;
}
export function syncProviderMatrix() {
return {
include: Object.values(providers).map((provider) => ({
provider: provider.id,
name: provider.name,
})),
};
}
async function readExisting(modelsDir: string) {
const existing = new Map<string, { text: string; toml: ExistingModel; symlink: boolean }>();
for (const { file, symlink } of await tomlFiles(modelsDir)) {
const text = await Bun.file(path.join(modelsDir, file)).text();
const parsed = ExistingModel.safeParse(Bun.TOML.parse(text));
if (!parsed.success) {
parsed.error.cause = { path: path.join(modelsDir, file) };
throw parsed.error;
}
existing.set(file, { text, toml: parsed.data, symlink });
}
return existing;
}
async function tomlFiles(root: string, dir = "") {
const result: Array<{ file: string; symlink: boolean }> = [];
for (const entry of await readdir(path.join(root, dir), { withFileTypes: true })) {
const file = path.join(dir, entry.name);
if (entry.isDirectory()) {
result.push(...await tomlFiles(root, file));
} else if (entry.name.endsWith(".toml") && (entry.isFile() || entry.isSymbolicLink())) {
result.push({ file, symlink: entry.isSymbolicLink() });
}
}
return result;
}
function summarize(
provider: { id: string; name: string },
files: SyncResult["files"],
unchanged: number,
notices: string[],
): SyncResult {
return {
id: provider.id,
name: provider.name,
status: files.length > 0 ? "changed" : "unchanged",
created: files.filter((file) => file.status === "created").length,
updated: files.filter((file) => file.status === "updated").length,
deleted: files.filter((file) => file.status === "deleted").length,
unchanged,
notices,
files,
};
}
function sameModel(
relativePath: string,
current: ExistingModel,
desired: z.infer<typeof SyncedAuthoredModel>,
) {
const parsed = SyncedAuthoredModel.safeParse({
id: relativePath.slice(0, -5),
...current,
});
return parsed.success && stable(parsed.data) === stable(desired);
}
function stable(value: unknown): string {
if (Array.isArray(value)) {
const items = value.map(stable);
const ordered = value.every((item) => item === null || typeof item !== "object")
? items.sort()
: items;
return `[${ordered.join(",")}]`;
}
if (value !== null && typeof value === "object") {
return `{${Object.entries(value)
.filter(([, item]) => item !== undefined)
.sort(([a], [b]) => a.localeCompare(b))
.map(([key, item]) => `${JSON.stringify(key)}:${stable(item)}`)
.join(",")}}`;
}
return JSON.stringify(value);
}
async function writeReport(target: string, results: SyncResult[]) {
await mkdir(".sync", { recursive: true });
const lines = [
`Updates model TOMLs for the \`${target}\` sync target.`,
"",
"| Provider | Status | Created | Updated | Deleted |",
"| --- | --- | ---: | ---: | ---: |",
];
for (const result of results) {
lines.push(
`| ${result.name} | ${result.status} | ${result.created} | ${result.updated} | ${result.deleted} |`,
);
}
for (const result of results.filter((item) => item.files.length > 0)) {
lines.push("", `<details><summary>${result.name} changed files</summary>`, "");
for (const file of result.files) {
lines.push(`- ${file.status}: \`${file.path}\``);
}
lines.push("", "</details>");
}
const noticeResults = results.filter((item) => item.notices.length > 0);
if (noticeResults.length > 0) {
lines.push("", "## Notices");
for (const result of noticeResults) {
lines.push("", `### ${result.name}`);
for (const notice of result.notices) {
lines.push(`- ${notice}`);
}
}
}
lines.push("", "This PR was created automatically by the daily model sync workflow.");
await Bun.write(".sync/model-sync-report.md", `${lines.join("\n")}\n`);
}
function quote(value: string) {
return `"${value.replaceAll("\\", "\\\\").replaceAll('"', '\\"')}"`;
}
function formatInteger(n: number) {
return String(n).replace(/\B(?=(\d{3})+(?!\d))/g, "_");
}
function formatNumber(n: number) {
return Number.isInteger(n) ? formatInteger(n) : String(n);
}
function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
const lines: string[] = [];
const extendsLines: string[] = [];
if ("extends" in model) {
extendsLines.push("[extends]");
extendsLines.push(`from = ${quote(model.extends.from)}`);
if (model.extends.omit !== undefined) {
extendsLines.push(`omit = [${model.extends.omit.map(quote).join(", ")}]`);
}
}
if (model.name !== undefined) lines.push(`name = ${quote(model.name)}`);
if (model.family !== undefined) lines.push(`family = ${quote(model.family)}`);
if (model.release_date !== undefined) lines.push(`release_date = ${quote(model.release_date)}`);
if (model.last_updated !== undefined) lines.push(`last_updated = ${quote(model.last_updated)}`);
if (model.attachment !== undefined) lines.push(`attachment = ${model.attachment}`);
if (model.reasoning !== undefined) lines.push(`reasoning = ${model.reasoning}`);
if (model.temperature !== undefined) lines.push(`temperature = ${model.temperature}`);
if (model.tool_call !== undefined) lines.push(`tool_call = ${model.tool_call}`);
if (model.structured_output !== undefined) {
lines.push(`structured_output = ${model.structured_output}`);
}
if (model.knowledge !== undefined) lines.push(`knowledge = ${quote(model.knowledge)}`);
if (model.open_weights !== undefined) lines.push(`open_weights = ${model.open_weights}`);
if (model.status !== undefined) lines.push(`status = ${quote(model.status)}`);
if (extendsLines.length > 0) {
if (lines.length > 0) lines.push("");
lines.push(...extendsLines);
}
if (model.interleaved !== undefined) {
lines.push("");
if (model.interleaved === true) {
lines.push("interleaved = true");
} else {
lines.push("[interleaved]");
lines.push(`field = ${quote(model.interleaved.field)}`);
}
}
if (model.cost !== undefined) {
lines.push("", "[cost]");
lines.push(`input = ${formatNumber(model.cost.input)}`);
lines.push(`output = ${formatNumber(model.cost.output)}`);
if (model.cost.reasoning !== undefined) {
lines.push(`reasoning = ${formatNumber(model.cost.reasoning)}`);
}
if (model.cost.cache_read !== undefined) {
lines.push(`cache_read = ${formatNumber(model.cost.cache_read)}`);
}
if (model.cost.cache_write !== undefined) {
lines.push(`cache_write = ${formatNumber(model.cost.cache_write)}`);
}
if (model.cost.input_audio !== undefined) {
lines.push(`input_audio = ${formatNumber(model.cost.input_audio)}`);
}
if (model.cost.output_audio !== undefined) {
lines.push(`output_audio = ${formatNumber(model.cost.output_audio)}`);
}
for (const tier of model.cost.tiers ?? []) {
lines.push("", "[[cost.tiers]]");
lines.push(`tier = { size = ${formatInteger(tier.tier.size)} }`);
lines.push(`input = ${formatNumber(tier.input)}`);
lines.push(`output = ${formatNumber(tier.output)}`);
if (tier.reasoning !== undefined) lines.push(`reasoning = ${formatNumber(tier.reasoning)}`);
if (tier.cache_read !== undefined) lines.push(`cache_read = ${formatNumber(tier.cache_read)}`);
if (tier.cache_write !== undefined) lines.push(`cache_write = ${formatNumber(tier.cache_write)}`);
}
}
if (model.limit !== undefined) {
lines.push("", "[limit]");
if (model.limit.context !== undefined) lines.push(`context = ${formatInteger(model.limit.context)}`);
if (model.limit.input !== undefined) lines.push(`input = ${formatInteger(model.limit.input)}`);
if (model.limit.output !== undefined) lines.push(`output = ${formatInteger(model.limit.output)}`);
}
if (model.modalities !== undefined) {
lines.push("", "[modalities]");
if (model.modalities.input !== undefined) {
lines.push(`input = [${model.modalities.input.map(quote).join(", ")}]`);
}
if (model.modalities.output !== undefined) {
lines.push(`output = [${model.modalities.output.map(quote).join(", ")}]`);
}
}
return `${lines.join("\n")}\n`;
}
export async function main(args = process.argv.slice(2)) {
if (args.includes("--list-providers")) {
console.log(JSON.stringify(syncProviderMatrix()));
return;
}
const target = args.find((arg) => !arg.startsWith("-")) ?? "aggregators";
const results = await syncTargets(target, {
dryRun: args.includes("--dry-run"),
newOnly: args.includes("--new-only"),
});
await writeReport(target, results);
console.log("\nSync summary");
for (const result of results) {
console.log(
`${result.name}: ${result.created} created, ${result.updated} updated, ${result.deleted} deleted`,
);
}
}
if (import.meta.main) await main();
@@ -0,0 +1,176 @@
import { z } from "zod";
import type { ExistingModel, SyncProvider } from "../index.js";
import {
buildOpenRouterModel,
OpenRouterModel,
OpenRouterResponse,
} from "./openrouter.js";
const API_BASE = "https://api.cloudflare.com/client/v4/accounts";
const CloudflareOpenRouterResponse = z.object({
result: z.union([OpenRouterResponse, z.array(OpenRouterModel)]).optional(),
result_info: z.object({
page: z.number().optional(),
total_pages: z.number().optional(),
}).passthrough().optional(),
}).passthrough();
const CloudflareModel = z.object({
id: z.string(),
name: z.string(),
created: z.number(),
hugging_face_id: z.string().nullable().optional(),
context_length: z.number(),
max_output_length: z.number().nullable().optional(),
input_modalities: z.array(z.string()).optional(),
output_modalities: z.array(z.string()).optional(),
pricing: z.object({
prompt: z.string(),
completion: z.string(),
internal_reasoning: z.string().optional(),
input_cache_read: z.string().optional(),
input_cache_write: z.string().optional(),
}),
supported_features: z.array(z.string()).optional(),
supported_sampling_parameters: z.array(z.string()).optional(),
}).passthrough();
const CloudflareResponse = z.object({
data: z.array(CloudflareModel),
}).passthrough();
type CloudflareModel = z.infer<typeof CloudflareModel>;
export const cloudflareWorkersAi = {
id: "cloudflare-workers-ai",
name: "Cloudflare Workers AI",
modelsDir: "providers/cloudflare-workers-ai/models",
async fetchModels() {
const accountID = process.env.CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID;
const token = process.env.CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN;
if (accountID === undefined || token === undefined) {
throw new Error(
"Cloudflare Workers AI sync requires CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID and CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN",
);
}
const first = await fetchPage(accountID, token, 1);
const models = parseCloudflareModels(first);
const pageInfo = CloudflareOpenRouterResponse.safeParse(first).success
? CloudflareOpenRouterResponse.parse(first).result_info
: undefined;
for (let page = 2; page <= (pageInfo?.total_pages ?? 1); page++) {
models.push(...parseCloudflareModels(await fetchPage(accountID, token, page)));
}
return { data: models };
},
parseModels(raw) {
return parseCloudflareModels(raw);
},
translateModel(model, context) {
const normalized = normalizeModel(model);
const id = normalized.id.replace(/^workers-ai\//, "");
return {
id,
model: buildWorkersAiModel(normalized, context.existing(id)),
};
},
} satisfies SyncProvider<CloudflareModel>;
function buildWorkersAiModel(model: z.infer<typeof OpenRouterModel>, existing: ExistingModel | undefined) {
const synced = buildOpenRouterModel(model, existing);
return {
...synced,
name: existing?.name ?? synced.name,
release_date: existing?.release_date ?? synced.release_date,
last_updated: existing?.last_updated ?? synced.last_updated,
limit: {
...synced.limit,
output: existing?.limit?.output ?? synced.limit.output,
},
};
}
async function fetchPage(accountID: string, token: string, page: number) {
const url = new URL(`${API_BASE}/${accountID}/ai/models/search`);
url.searchParams.set("format", "openrouter");
url.searchParams.set("per_page", "1000");
url.searchParams.set("page", String(page));
const response = await fetch(url, {
headers: { Authorization: `Bearer ${token}` },
});
if (!response.ok) {
throw new Error(
`Cloudflare Workers AI models request failed: ${response.status} ${response.statusText}${await responseDetails(response)}`,
);
}
return response.json();
}
function parseCloudflareModels(raw: unknown) {
const cloudflare = CloudflareResponse.safeParse(raw);
if (cloudflare.success) return cloudflare.data.data;
const direct = OpenRouterResponse.safeParse(raw);
if (direct.success) return direct.data.data;
const wrapped = CloudflareOpenRouterResponse.parse(raw);
if (wrapped.result === undefined) {
throw new Error("Cloudflare Workers AI response did not include model data");
}
return Array.isArray(wrapped.result) ? wrapped.result : wrapped.result.data;
}
function normalizeModel(model: CloudflareModel) {
if ("architecture" in model && "top_provider" in model && "supported_parameters" in model) {
return OpenRouterModel.parse(model);
}
return OpenRouterModel.parse({
id: model.id.startsWith("@cf/") ? model.id : `@cf/${model.id.replace(/^@cf\//, "")}`,
name: model.name,
created: model.created,
hugging_face_id: model.hugging_face_id ?? null,
knowledge_cutoff: null,
context_length: model.context_length,
architecture: {
input_modalities: model.input_modalities ?? ["text"],
output_modalities: model.output_modalities ?? ["text"],
},
pricing: model.pricing,
top_provider: {
context_length: model.context_length,
max_completion_tokens: model.max_output_length ?? null,
},
supported_parameters: [
...model.supported_sampling_parameters ?? [],
...model.supported_features ?? [],
],
});
}
async function responseDetails(response: Response) {
const text = await response.text();
if (text.length === 0) return "";
try {
const body = z.object({
errors: z.array(z.object({
code: z.union([z.string(), z.number()]).optional(),
message: z.string().optional(),
}).passthrough()).optional(),
}).passthrough().parse(JSON.parse(text));
const details = body.errors
?.map((error) => [error.code, error.message].filter(Boolean).join(": "))
.filter((message) => message.length > 0)
.join("; ");
return details === undefined || details.length === 0 ? "" : ` (${details})`;
} catch {
return "";
}
}
+138
View File
@@ -0,0 +1,138 @@
import { z } from "zod";
import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js";
const API_ENDPOINT = "https://generativelanguage.googleapis.com/v1beta/models";
const GoogleModel = z.object({
name: z.string(),
baseModelId: z.string().optional(),
version: z.string().optional(),
displayName: z.string().optional(),
description: z.string().optional(),
inputTokenLimit: z.number().int().nonnegative(),
outputTokenLimit: z.number().int().nonnegative(),
supportedGenerationMethods: z.array(z.string()).optional(),
temperature: z.number().optional(),
topP: z.number().optional(),
topK: z.number().optional(),
maxTemperature: z.number().optional(),
thinking: z.boolean().optional(),
}).passthrough();
const GoogleResponse = z.object({
models: z.array(GoogleModel).optional(),
nextPageToken: z.string().optional(),
}).passthrough();
type GoogleModel = z.infer<typeof GoogleModel>;
export const google = {
id: "google",
name: "Google",
modelsDir: "providers/google/models",
skipCreates: true,
sourceID(model) {
return model.name.replace(/^models\//, "");
},
skippedNotice(ids) {
if (ids.length === 0) return [];
return [
`${ids.length} Google models returned by the API were not created because the Models API does not provide authoritative modalities, pricing, knowledge cutoff, release date, tool calling, or structured output metadata. Existing models are still updated from API-authoritative fields.`,
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
];
},
async fetchModels() {
const key = process.env.GOOGLE_API_KEY
?? process.env.GEMINI_API_KEY
?? process.env.GOOGLE_GENERATIVE_AI_API_KEY;
if (key === undefined) {
throw new Error("Google sync requires GOOGLE_API_KEY, GEMINI_API_KEY, or GOOGLE_GENERATIVE_AI_API_KEY");
}
const models: GoogleModel[] = [];
let pageToken: string | undefined;
do {
const url = new URL(API_ENDPOINT);
url.searchParams.set("key", key);
url.searchParams.set("pageSize", "1000");
if (pageToken !== undefined) url.searchParams.set("pageToken", pageToken);
const response = await fetch(url);
if (!response.ok) {
throw new Error(`Google models request failed: ${response.status} ${response.statusText}`);
}
const page = GoogleResponse.parse(await response.json());
models.push(...page.models ?? []);
pageToken = page.nextPageToken;
} while (pageToken !== undefined);
return { models };
},
parseModels(raw) {
return GoogleResponse.parse(raw).models ?? [];
},
translateModel(model, context) {
const id = model.name.replace(/^models\//, "");
const existing = context.existing(id);
if (existing === undefined) return undefined;
return {
id,
model: buildModel(model, existing),
};
},
} satisfies SyncProvider<GoogleModel>;
function buildModel(model: GoogleModel, existing: ExistingModel): SyncedModel {
const name = existing.name;
const releaseDate = existing.release_date;
const lastUpdated = existing.last_updated;
const attachment = existing.attachment;
const reasoning = existing.reasoning;
const toolCall = existing.tool_call;
const openWeights = existing.open_weights;
const limit = existing.limit;
const modalities = existing.modalities;
if (
name === undefined
|| releaseDate === undefined
|| lastUpdated === undefined
|| attachment === undefined
|| reasoning === undefined
|| toolCall === undefined
|| openWeights === undefined
|| limit === undefined
|| modalities === undefined
) {
throw new Error(`Google model ${model.name} has incomplete local TOML metadata required for sync`);
}
return {
name: model.displayName ?? name,
family: existing.family,
release_date: releaseDate,
last_updated: lastUpdated,
attachment,
reasoning: model.thinking ?? reasoning,
temperature: model.temperature !== undefined || model.maxTemperature !== undefined
? true
: existing.temperature,
tool_call: toolCall,
structured_output: existing.structured_output,
knowledge: existing.knowledge,
open_weights: openWeights,
status: existing.status,
interleaved: existing.interleaved,
cost: existing.cost,
limit: {
input: limit.input,
context: model.inputTokenLimit,
output: model.outputTokenLimit,
},
modalities,
};
}
@@ -0,0 +1,334 @@
import { z } from "zod";
import { readFileSync, readdirSync } from "node:fs";
import path from "node:path";
import { ModelFamilyValues } from "../../family.js";
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
const API_ENDPOINT = "https://openrouter.ai/api/v1/models";
const PROVIDERS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "providers");
const modelFilesByProvider = new Map<string, Set<string>>();
const canonicalTomlByModel = new Map<string, Record<string, unknown>>();
const CANONICAL_PROVIDER_PREFIXES = {
anthropic: "anthropic",
cohere: "cohere",
deepseek: "deepseek",
google: "google",
meta: "llama",
"meta-llama": "llama",
minimax: "minimax",
mistralai: "mistral",
moonshotai: "moonshotai",
openai: "openai",
"x-ai": "xai",
xai: "xai",
xiaomi: "xiaomi",
zai: "zai",
"z-ai": "zai",
} as const;
export const OpenRouterModel = z.object({
id: z.string(),
name: z.string(),
created: z.number(),
hugging_face_id: z.string().nullable(),
knowledge_cutoff: z.string().nullable(),
context_length: z.number(),
architecture: z.object({
input_modalities: z.array(z.string()),
output_modalities: z.array(z.string()),
}),
pricing: z.object({
prompt: z.string(),
completion: z.string(),
internal_reasoning: z.string().optional(),
input_cache_read: z.string().optional(),
input_cache_write: z.string().optional(),
}),
top_provider: z.object({
context_length: z.number().nullable(),
max_completion_tokens: z.number().nullable(),
}),
supported_parameters: z.array(z.string()),
});
export const OpenRouterResponse = z.object({
data: z.array(OpenRouterModel),
}).passthrough();
export type OpenRouterModel = z.infer<typeof OpenRouterModel>;
export const openrouter = {
id: "openrouter",
name: "OpenRouter",
modelsDir: "providers/openrouter/models",
async fetchModels() {
const headers = process.env.OPENROUTER_API_KEY
? { Authorization: `Bearer ${process.env.OPENROUTER_API_KEY}` }
: undefined;
const response = await fetch(API_ENDPOINT, { headers });
if (!response.ok) {
throw new Error(`OpenRouter request failed: ${response.status} ${response.statusText}`);
}
return response.json();
},
parseModels(raw) {
return OpenRouterResponse.parse(raw).data;
},
translateModel(model, context) {
return {
id: model.id,
model: buildOpenRouterModel(model, context.existing(model.id)),
};
},
} satisfies SyncProvider<OpenRouterModel>;
function dateFromTimestamp(timestamp: number) {
return new Date(timestamp * 1000).toISOString().slice(0, 10);
}
function price(value: string | undefined) {
if (value === undefined) return undefined;
const number = Number(value);
return Number.isFinite(number) && number >= 0
? Math.round(number * 1_000_000_000_000) / 1_000_000
: undefined;
}
type Modality = "text" | "audio" | "image" | "video" | "pdf";
function modalities(values: string[], fallback: Modality[]): Modality[] {
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
const result = values
.map((value) => value.toLowerCase())
.map((value) => value === "file" ? "pdf" : value)
.filter((value): value is Modality => allowed.has(value as Modality));
return [...new Set(result.length > 0 ? result : fallback)];
}
function inferFamily(model: OpenRouterModel, name: string) {
const target = `${model.id} ${name}`.toLowerCase();
return [...ModelFamilyValues]
.sort((a, b) => b.length - a.length)
.find((family) => {
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
if (family === "o") {
return new RegExp(`(^|[^a-z0-9])${value}(?=\\d|$|[^a-z0-9])`).test(target);
}
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
});
}
export function buildOpenRouterModel(model: OpenRouterModel, existing: ExistingModel | undefined): SyncedModel {
const params = new Set(model.supported_parameters);
const name = model.name.replace(/^[^:]+:\s+/, "");
const input = modalities(model.architecture.input_modalities, ["text"]);
const output = modalities(model.architecture.output_modalities, ["text"]);
const prompt = price(model.pricing.prompt);
const completion = price(model.pricing.completion);
const reasoning = params.has("reasoning") || params.has("include_reasoning");
const context = model.top_provider.context_length ?? model.context_length;
const family = inferFamily(model, name);
const releaseDate = dateFromTimestamp(model.created);
const familyValue = existing?.family === "o" && family !== "o"
? family
: (existing?.family ?? family);
const attachment = input.some((value) => value !== "text");
const toolCall = params.has("tools") || params.has("tool_choice");
const structuredOutput = params.has("structured_outputs");
const knowledge = model.knowledge_cutoff?.slice(0, 10) ?? existing?.knowledge;
const openWeights = Boolean(model.hugging_face_id);
const cost = prompt !== undefined && completion !== undefined
? {
input: prompt,
output: completion,
reasoning: reasoning ? price(model.pricing.internal_reasoning) : undefined,
cache_read: price(model.pricing.input_cache_read),
cache_write: price(model.pricing.input_cache_write),
tiers: existing?.cost?.tiers,
}
: existing?.cost;
const limit = {
context,
input: existing?.limit?.input,
output: model.top_provider.max_completion_tokens ?? existing?.limit?.output ?? context,
};
const canonical = resolveCanonicalModel(model.id);
if (canonical !== undefined) {
return {
extends: {
from: canonical.from,
omit: canonicalOmit(canonical.provider, canonical.modelID, cost, limit),
},
...canonicalRuntimeOverrides(canonical.provider, canonical.modelID, {
name: model.id.endsWith(":free") ? name : undefined,
attachment,
reasoning,
}),
temperature: params.has("temperature"),
tool_call: toolCall,
structured_output: structuredOutput,
status: existing?.status,
interleaved: existing?.interleaved,
cost,
limit,
modalities: { input, output },
};
}
return {
name,
family: familyValue,
release_date: releaseDate,
last_updated: releaseDate,
attachment,
reasoning,
temperature: params.has("temperature"),
tool_call: toolCall,
structured_output: structuredOutput,
knowledge,
open_weights: openWeights,
status: existing?.status,
interleaved: existing?.interleaved,
cost,
limit,
modalities: { input, output },
} satisfies SyncedFullModel;
}
function resolveCanonicalModel(openrouterID: string) {
const [prefix, ...modelParts] = openrouterID.split("/");
if (prefix === undefined || modelParts.length === 0) return undefined;
if (openrouterID.startsWith("~/") || prefix.startsWith("~")) return undefined;
const provider = CANONICAL_PROVIDER_PREFIXES[prefix as keyof typeof CANONICAL_PROVIDER_PREFIXES];
if (provider === undefined) return undefined;
const modelID = modelParts.join("/").replace(/:free$/, "");
const candidates = canonicalCandidates(provider, modelID);
const match = candidates.find((candidate) => {
return canonicalModelExists(provider, candidate);
});
return match === undefined
? undefined
: {
from: `${provider}/${match}`,
provider,
modelID: match,
};
}
function canonicalModelExists(provider: string, modelID: string) {
let files = modelFilesByProvider.get(provider);
if (files === undefined) {
try {
files = new Set(readdirSync(path.join(PROVIDERS_DIR, provider, "models")));
} catch {
files = new Set();
}
modelFilesByProvider.set(provider, files);
}
return files.has(`${modelID}.toml`);
}
function canonicalOmit(
provider: string,
modelID: string,
cost: SyncedFullModel["cost"],
limit: SyncedFullModel["limit"],
) {
const toml = canonicalToml(provider, modelID);
const omit = ["provider", "experimental"].filter((key) => toml[key] !== undefined);
const baseCost = toml.cost;
if (baseCost !== undefined && baseCost !== null && typeof baseCost === "object" && !Array.isArray(baseCost)) {
if (cost === undefined) {
omit.push("cost");
} else {
for (const key of ["reasoning", "cache_read", "cache_write", "input_audio", "output_audio", "tiers"] as const) {
if ((baseCost as Record<string, unknown>)[key] !== undefined && cost[key] === undefined) {
omit.push(`cost.${key}`);
}
}
if (hasLegacyContextOver200k(baseCost) && cost.tiers === undefined) {
omit.push("cost.context_over_200k");
}
}
}
const baseLimit = toml.limit;
if (
baseLimit !== undefined &&
baseLimit !== null &&
typeof baseLimit === "object" &&
!Array.isArray(baseLimit) &&
(baseLimit as Record<string, unknown>).input !== undefined &&
limit.input === undefined
) {
omit.push("limit.input");
}
return omit.length > 0 ? omit : undefined;
}
function hasLegacyContextOver200k(cost: object) {
const tiers = (cost as { tiers?: unknown }).tiers;
if (!Array.isArray(tiers) || tiers.length !== 1) return false;
const tier = tiers[0];
if (tier === null || typeof tier !== "object" || Array.isArray(tier)) return false;
const tierConfig = (tier as { tier?: unknown }).tier;
if (tierConfig === null || typeof tierConfig !== "object" || Array.isArray(tierConfig)) return false;
const size = (tierConfig as { size?: unknown }).size;
return typeof size === "number" && size >= 200_000;
}
function canonicalRuntimeOverrides(
provider: string,
modelID: string,
values: Pick<SyncedFullModel, "name" | "attachment" | "reasoning">,
) {
const toml = canonicalToml(provider, modelID);
return Object.fromEntries(
Object.entries(values).filter(([key, value]) => value !== undefined && toml[key] !== value),
);
}
function canonicalToml(provider: string, modelID: string) {
const key = `${provider}/${modelID}`;
let toml = canonicalTomlByModel.get(key);
if (toml === undefined) {
const filePath = path.join(PROVIDERS_DIR, provider, "models", `${modelID}.toml`);
toml = Bun.TOML.parse(readFileSync(filePath, "utf8")) as Record<string, unknown>;
canonicalTomlByModel.set(key, toml);
}
return toml;
}
function canonicalCandidates(provider: string, modelID: string) {
const candidates = [modelID];
if (provider === "anthropic") {
candidates.push(modelID.replace(/(claude-(?:opus|sonnet|haiku)-\d+)\.(\d+)/, "$1-$2"));
candidates.push(modelID.replace(/^claude-3\.5-/, "claude-3-5-"));
}
if (provider === "llama") {
candidates.push(modelID.replace(/^llama-(\d+)-(\d+)/, "llama-$1.$2"));
candidates.push(modelID.replace(/^llama-(4)-(maverick|scout)$/, "llama-$1-$2-17b"));
}
if (provider === "mistral") {
candidates.push(modelID.replace(/-latest$/, ""));
}
if (provider === "minimax") {
candidates.push(modelID.replace(/^minimax-m/, "MiniMax-M"));
}
return [...new Set(candidates)];
}
+211
View File
@@ -0,0 +1,211 @@
import { z } from "zod";
import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js";
const API_BASE = "https://api.x.ai/v1";
const XAIModel = z.object({
id: z.string(),
canonical_id: z.string().optional(),
created: z.number().int().nonnegative(),
aliases: z.array(z.string()).optional(),
input_modalities: z.array(z.string()).optional(),
output_modalities: z.array(z.string()).optional(),
prompt_text_token_price: z.number().int().nonnegative().optional(),
cached_prompt_text_token_price: z.number().int().nonnegative().optional(),
completion_text_token_price: z.number().int().nonnegative().optional(),
max_prompt_length: z.number().int().nonnegative().optional(),
}).passthrough();
const XAIModelList = z.object({
models: z.array(XAIModel),
}).passthrough();
const XAIResponse = z.object({
models: z.array(XAIModel),
});
const XAIAPIKey = z.object({
acls: z.array(z.string()),
}).passthrough();
type XAIModel = z.infer<typeof XAIModel>;
export const xai = {
id: "xai",
name: "xAI",
modelsDir: "providers/xai/models",
skipCreates: true,
sourceID(model) {
return model.id;
},
skippedNotice(ids) {
if (ids.length === 0) return [];
return [
`${ids.length} xAI models returned by the API were not created because the Models API does not provide enough authoritative metadata for the catalog, especially output token limits and some feature/capability flags. Existing models are still updated from API-authoritative fields.`,
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
];
},
async fetchModels() {
const key = process.env.XAI_API_KEY;
if (key === undefined) throw new Error("xAI sync requires XAI_API_KEY");
await assertFullModelAccess(key);
const models = await Promise.all([
fetchTypedModels(key, "language-models"),
fetchTypedModels(key, "image-generation-models"),
fetchTypedModels(key, "video-generation-models"),
]);
return { models: models.flat() };
},
parseModels(raw) {
const models = XAIResponse.parse(raw).models;
const seen = new Set<string>();
const expanded: XAIModel[] = [];
for (const model of models) {
if (!seen.has(model.id)) {
seen.add(model.id);
expanded.push(model);
}
}
for (const model of models) {
for (const alias of model.aliases ?? []) {
if (seen.has(alias)) continue;
seen.add(alias);
expanded.push({ ...model, id: alias, canonical_id: model.id });
}
}
return expanded;
},
translateModel(model, context) {
const existing = context.existing(model.id);
if (existing === undefined) return undefined;
return {
id: model.id,
model: buildModel(model, existing),
};
},
} satisfies SyncProvider<XAIModel>;
async function assertFullModelAccess(key: string) {
const response = await fetch(`${API_BASE}/api-key`, {
headers: { Authorization: `Bearer ${key}` },
});
if (!response.ok) {
throw new Error(`xAI API key metadata request failed: ${response.status} ${response.statusText}`);
}
const apiKey = XAIAPIKey.parse(await response.json());
if (!apiKey.acls.includes("api-key:model:*")) {
throw new Error("xAI sync requires XAI_API_KEY to include api-key:model:* so the model list is not ACL-filtered");
}
}
async function fetchTypedModels(key: string, endpoint: string) {
const response = await fetch(`${API_BASE}/${endpoint}`, {
headers: { Authorization: `Bearer ${key}` },
});
if (!response.ok) {
throw new Error(`xAI ${endpoint} request failed: ${response.status} ${response.statusText}`);
}
return XAIModelList.parse(await response.json()).models;
}
function dateFromTimestamp(timestamp: number) {
return new Date(timestamp * 1000).toISOString().slice(0, 10);
}
type Modality = "text" | "audio" | "image" | "video" | "pdf";
function modalities(values: string[] | undefined, fallback: Modality[]) {
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
const result = (values ?? [])
.map((value) => value.toLowerCase())
.filter((value): value is Modality => allowed.has(value as Modality));
if (result.includes("image")) result.push("pdf");
return [...new Set(result.length > 0 ? result : fallback)];
}
function tokenPrice(value: number | undefined) {
if (value === undefined) return undefined;
return value / 10_000;
}
function preservedCostTiers(existing: ExistingModel) {
// The xAI models API exposes base pricing only; long-context tiers are curated from xAI docs/console.
return existing.cost?.tiers;
}
function cost(model: XAIModel, existing: ExistingModel) {
const input = tokenPrice(model.prompt_text_token_price);
const output = tokenPrice(model.completion_text_token_price);
if (input === undefined || output === undefined) return existing.cost;
return {
input,
output,
reasoning: existing.cost?.reasoning,
cache_read: tokenPrice(model.cached_prompt_text_token_price),
cache_write: existing.cost?.cache_write,
input_audio: existing.cost?.input_audio,
output_audio: existing.cost?.output_audio,
tiers: preservedCostTiers(existing),
};
}
function buildModel(model: XAIModel, existing: ExistingModel): SyncedModel {
const name = existing.name;
const attachment = existing.attachment;
const reasoning = existing.reasoning;
const toolCall = existing.tool_call;
const openWeights = existing.open_weights;
const limit = existing.limit;
const releaseDate = existing.release_date;
const lastUpdated = existing.last_updated;
if (
name === undefined
|| attachment === undefined
|| reasoning === undefined
|| toolCall === undefined
|| openWeights === undefined
|| limit === undefined
|| (model.canonical_id !== undefined && releaseDate === undefined)
|| (model.canonical_id !== undefined && lastUpdated === undefined)
) {
throw new Error(`xAI model ${model.id} has incomplete local TOML metadata required for sync`);
}
const input = modalities(model.input_modalities, existing.modalities?.input ?? ["text"]);
const output = modalities(model.output_modalities, existing.modalities?.output ?? ["text"]);
const created = dateFromTimestamp(model.created);
return {
name,
family: existing.family,
release_date: model.canonical_id === undefined ? created : releaseDate!,
last_updated: model.canonical_id === undefined ? created : lastUpdated!,
attachment: input.some((value) => value !== "text"),
reasoning,
temperature: existing.temperature,
tool_call: toolCall,
structured_output: existing.structured_output,
knowledge: existing.knowledge,
open_weights: openWeights,
status: existing.status,
interleaved: existing.interleaved,
cost: cost(model, existing),
limit: {
input: limit.input,
context: model.max_prompt_length ?? limit.context,
output: limit.output,
},
modalities: { input, output },
};
}
+2 -1
View File
@@ -15,7 +15,8 @@ output = 25.000
cache_read = 0.500
cache_write = 6.250
[cost.context_over_200k]
[[cost.tiers]]
tier = { size = 200_000 }
input = 10.000
output = 37.500
cache_read = 1.000
+2 -1
View File
@@ -16,7 +16,8 @@ output = 180.000
cache_read = 0
cache_write = 0
[cost.context_over_200k]
[[cost.tiers]]
tier = { size = 272_000 }
input = 60.000
output = 270.000
+2 -1
View File
@@ -16,7 +16,8 @@ output = 15.000
cache_read = 0.250
cache_write = 0
[cost.context_over_200k]
[[cost.tiers]]
tier = { size = 272_000 }
input = 5.000
output = 22.500
@@ -1,5 +1,5 @@
name = "DeepSeek V4 Flash Think"
family = "deepseek"
name = "DeepSeek V4 Flash (Alibaba Cloud)"
family = "deepseek-flash"
release_date = "2026-04-24"
last_updated = "2026-04-24"
attachment = false
@@ -14,9 +14,9 @@ open_weights = true
field = "reasoning_content"
[cost]
input = 0.154
output = 0.308
cache_read = 0.0308
input = 0.14
output = 0.28
cache_read = 0.028
[limit]
context = 1_000_000
@@ -1,25 +1,27 @@
name = "DeepSeek R1 TEE"
name = "DeepSeek V4 Pro (Alibaba Cloud)"
family = "deepseek-thinking"
release_date = "2025-12-29"
last_updated = "2026-01-10"
release_date = "2026-04-24"
last_updated = "2026-04-24"
attachment = false
reasoning = true
temperature = true
tool_call = false
tool_call = true
structured_output = true
knowledge = "2025-05"
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.30
output = 1.20
input = 1.69
output = 3.38
cache_read = 0.13
[limit]
context = 163_840
output = 163_840
context = 1_000_000
output = 384_000
[modalities]
input = ["text"]
output = ["text"]
[interleaved]
field = "reasoning_content"
@@ -1,7 +1,7 @@
name = "GLM 4.5 TEE"
name = "GLM-5.1 (Alibaba Cloud)"
family = "glm"
release_date = "2025-12-29"
last_updated = "2026-01-10"
release_date = "2026-03-27"
last_updated = "2026-03-27"
attachment = false
reasoning = true
temperature = true
@@ -9,17 +9,19 @@ tool_call = true
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.35
output = 1.55
input = 0.84
output = 3.38
cache_read = 0.169
cache_write = 1.05625
[limit]
context = 131_072
output = 65_536
context = 200_000
output = 128_000
[modalities]
input = ["text"]
output = ["text"]
[interleaved]
field = "reasoning_content"
@@ -1,4 +1,4 @@
name = "Claude Opus 4.6"
name = "Claude Opus 4.6 Thinking"
family = "claude-opus"
release_date = "2026-02-05"
last_updated = "2026-03-13"
@@ -6,18 +6,29 @@ attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-05-31"
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 5
output = 25
cache_read = 0.5
cache_write = 6.25
[[cost.tiers]]
tier = { size = 200_000 }
input = 10
output = 37.5
cache_read = 1.0
cache_write = 12.5
[limit]
context = 200_000
output = 32_000
context = 1_000_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
@@ -6,15 +6,25 @@ attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-05-31"
open_weights = false
interleaved = true
[cost]
input = 5
output = 25
cache_read = 0.5
cache_write = 6.25
[[cost.tiers]]
tier = { size = 200_000 }
input = 10
output = 37.5
cache_read = 1.0
cache_write = 12.5
[limit]
context = 1_000_000
output = 128_000
@@ -0,0 +1,35 @@
name = "Claude Opus 4.7 Thinking"
family = "claude-opus"
release_date = "2026-04-16"
last_updated = "2026-04-16"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-01-31"
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 5
output = 25
cache_read = 0.5
cache_write = 6.25
[[cost.tiers]]
tier = { size = 200_000 }
input = 10
output = 37.5
cache_read = 1.0
cache_write = 12.5
[limit]
context = 1_000_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
@@ -6,15 +6,25 @@ attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-01-31"
open_weights = false
interleaved = true
[cost]
input = 5
output = 25
cache_read = 0.5
cache_write = 6.25
[[cost.tiers]]
tier = { size = 200_000 }
input = 10
output = 37.5
cache_read = 1.0
cache_write = 12.5
[limit]
context = 1_000_000
output = 128_000
@@ -1,28 +1,33 @@
name = "Claude Sonnet 4.6 Think"
name = "Claude Sonnet 4.6 Thinking"
family = "claude-sonnet"
release_date = "2026-02-17"
last_updated = "2026-02-17"
last_updated = "2026-03-13"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 3.00
output = 15.00
cache_read = 0.30
cache_write = 3.75
[cost.context_over_200k]
[[cost.tiers]]
tier = { size = 200_000 }
input = 6.00
output = 22.50
cache_read = 0.60
cache_write = 7.50
[limit]
context = 200_000
context = 1_000_000
output = 64_000
[modalities]
@@ -1,28 +1,32 @@
name = "Claude Sonnet 4.6"
family = "claude-sonnet"
release_date = "2026-02-17"
last_updated = "2026-02-17"
last_updated = "2026-03-13"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
interleaved = true
[cost]
input = 3.00
output = 15.00
cache_read = 0.30
cache_write = 3.75
[cost.context_over_200k]
[[cost.tiers]]
tier = { size = 200_000 }
input = 6.00
output = 22.50
cache_read = 0.60
cache_write = 7.50
[limit]
context = 200_000
context = 1_000_000
output = 64_000
[modalities]
@@ -1,24 +0,0 @@
name = "Coding-GLM-5-Free"
family = "glm"
release_date = "2026-02-11"
last_updated = "2026-02-11"
attachment = false
reasoning = true
temperature = true
tool_call = true
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.0
output = 0.0
[limit]
context = 204800
output = 131072
[modalities]
input = ["text"]
output = ["text"]
@@ -1,7 +1,7 @@
name = "Qwen3 235B A22B"
family = "qwen"
release_date = "2025-12-29"
last_updated = "2026-01-10"
name = "Coding GLM 5.1 (free)"
family = "glm-free"
release_date = "2026-04-11"
last_updated = "2026-04-11"
attachment = false
reasoning = true
temperature = true
@@ -9,17 +9,17 @@ tool_call = true
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.30
output = 1.20
input = 0
output = 0
[limit]
context = 40_960
output = 40_960
context = 200_000
output = 128_000
[modalities]
input = ["text"]
output = ["text"]
[interleaved]
field = "reasoning_content"
@@ -1,4 +1,4 @@
name = "Coding-GLM-5.1"
name = "Coding GLM 5.1"
family = "glm"
release_date = "2026-04-11"
last_updated = "2026-04-11"
@@ -6,7 +6,8 @@ attachment = false
reasoning = true
temperature = true
tool_call = true
open_weights = false
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
@@ -14,10 +15,11 @@ field = "reasoning_content"
[cost]
input = 0.06
output = 0.22
cache_read = 0.013
[limit]
context = 200000
output = 128000
context = 200_000
output = 128_000
[modalities]
input = ["text"]
@@ -1,20 +1,24 @@
name = "Coding-MiniMax-M2.7-Free"
family = "minimax"
name = "Coding MiniMax M2.7 (Free)"
family = "minimax-free"
release_date = "2026-03-18"
last_updated = "2026-03-18"
attachment = false
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0
output = 0
[limit]
context = 204_800
output = 13_100
output = 128_100
[modalities]
input = ["text"]
@@ -1,7 +1,7 @@
name = "Hermes 4 70B"
family = "nousresearch"
release_date = "2025-12-29"
last_updated = "2026-01-10"
name = "Coding MiniMax M2.7 Highspeed"
family = "minimax"
release_date = "2026-03-18"
last_updated = "2026-03-18"
attachment = false
reasoning = true
temperature = true
@@ -9,17 +9,17 @@ tool_call = true
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.11
output = 0.38
input = 0.2
output = 0.2
[limit]
context = 131_072
output = 131_072
context = 204_800
output = 128_100
[modalities]
input = ["text"]
output = ["text"]
[interleaved]
field = "reasoning_content"
@@ -1,7 +1,7 @@
name = "MiniMax M2.1 TEE"
name = "Coding MiniMax M2.7"
family = "minimax"
release_date = "2025-12-29"
last_updated = "2026-01-27"
release_date = "2026-03-18"
last_updated = "2026-03-18"
attachment = false
reasoning = true
temperature = true
@@ -9,17 +9,17 @@ tool_call = true
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.27
output = 1.12
input = 0.2
output = 0.2
[limit]
context = 196_608
output = 65_536
context = 204_800
output = 128_100
[modalities]
input = ["text"]
output = ["text"]
[interleaved]
field = "reasoning_content"
@@ -0,0 +1,17 @@
name = "Coding Xiaomi MiMo-V2.5-Pro"
family = "mimo-v2.5-pro"
last_updated = "2026-05-13"
[extends]
from = "xiaomi/mimo-v2.5-pro"
[cost]
input = 0.20
output = 0.60
cache_read = 0.04
[[cost.tiers]]
tier = { size = 256_000 }
input = 0.40
output = 1.20
cache_read = 0.08
@@ -0,0 +1,17 @@
name = "Coding Xiaomi MiMo-V2.5"
family = "mimo-v2.5"
last_updated = "2026-05-13"
[extends]
from = "xiaomi/mimo-v2.5"
[cost]
input = 0.08
output = 0.40
cache_read = 0.016
[[cost.tiers]]
tier = { size = 256_000 }
input = 0.16
output = 0.80
cache_read = 0.032
@@ -1,4 +1,4 @@
name = "DeepSeek V4 Flash"
name = "DeepSeek V4 Flash (DeepSeek)"
family = "deepseek-flash"
release_date = "2026-04-24"
last_updated = "2026-04-24"
@@ -1,4 +1,4 @@
name = "DeepSeek V4 Pro"
name = "DeepSeek V4 Pro (DeepSeek)"
family = "deepseek-thinking"
release_date = "2026-04-24"
last_updated = "2026-04-24"
@@ -0,0 +1,38 @@
name = "Doubao Seed 2.0 Code Preview"
family = "seed"
release_date = "2026-02-14"
last_updated = "2026-02-14"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.48
output = 2.41
cache_read = 0.09644
[[cost.tiers]]
tier = { size = 32_000 }
input = 0.72
output = 3.62
cache_read = 0.144656
[[cost.tiers]]
tier = { size = 128_000 }
input = 1.45
output = 7.23
cache_read = 0.28932
[limit]
context = 256_000
output = 128_000
[modalities]
input = ["text", "image", "video"]
output = ["text"]
@@ -0,0 +1,41 @@
name = "Doubao Seed 2.0 Lite 260428"
family = "seed"
release_date = "2026-04-28"
last_updated = "2026-04-28"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.08
output = 0.51
cache_read = 0.01692
input_audio = 1.269
[[cost.tiers]]
tier = { size = 32_000 }
input = 0.13
output = 0.76
cache_read = 0.02536
input_audio = 1.902
[[cost.tiers]]
tier = { size = 128_000 }
input = 0.25
output = 1.52
cache_read = 0.05072
input_audio = 3.804
[limit]
context = 256_000
output = 128_000
[modalities]
input = ["text", "image", "video"]
output = ["text"]
@@ -0,0 +1,41 @@
name = "Doubao Seed 2.0 Mini 260428"
family = "seed"
release_date = "2026-04-28"
last_updated = "2026-04-28"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.03
output = 0.28
cache_read = 0.00564
input_audio = 0.423
[[cost.tiers]]
tier = { size = 32_000 }
input = 0.06
output = 0.56
cache_read = 0.01128
input_audio = 0.846
[[cost.tiers]]
tier = { size = 128_000 }
input = 0.11
output = 1.13
cache_read = 0.02256
input_audio = 1.692
[limit]
context = 256_000
output = 128_000
[modalities]
input = ["text", "image", "video"]
output = ["text"]
@@ -0,0 +1,38 @@
name = "Doubao Seed 2.0 Pro"
family = "seed"
release_date = "2026-02-14"
last_updated = "2026-02-14"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.48
output = 2.41
cache_read = 0.09644
[[cost.tiers]]
tier = { size = 32_000 }
input = 0.72
output = 3.62
cache_read = 0.144656
[[cost.tiers]]
tier = { size = 128_000 }
input = 1.45
output = 7.23
cache_read = 0.28932
[limit]
context = 256_000
output = 128_000
[modalities]
input = ["text", "image", "video"]
output = ["text"]
@@ -12,8 +12,9 @@ open_weights = false
[cost]
input = 0.3
output = 2.499
output = 2.50
cache_read = 0.03
input_audio = 1.00
[limit]
context = 1_048_576
@@ -15,6 +15,12 @@ input = 1.25
output = 10
cache_read = 0.125
[[cost.tiers]]
tier = { size = 200_000 }
input = 2.50
output = 15.00
cache_read = 0.25
[limit]
context = 1_048_576
output = 65_536
@@ -15,6 +15,12 @@ input = 0.5
output = 3
cache_read = 0.05
[[cost.tiers]]
tier = { size = 200_000 }
input = 0.50
output = 3.00
cache_read = 0.05
[limit]
context = 1_048_576
output = 65_536
@@ -1,7 +1,7 @@
name = "Gemini 3.1 Flash Lite Preview"
name = "Gemini 3.1 Flash Lite"
family = "gemini-flash-lite"
release_date = "2026-03-03"
last_updated = "2026-03-03"
release_date = "2026-05-07"
last_updated = "2026-05-07"
attachment = true
reasoning = true
temperature = true
@@ -13,7 +13,8 @@ open_weights = false
[cost]
input = 0.25
output = 1.5
cache_read = 0.25
cache_read = 0.025
cache_write = 1.00
[limit]
context = 1_048_576
@@ -1,19 +1,25 @@
name = "Gemini 2.5 Pro Preview 05-06"
name = "Gemini 3.1 Pro Preview Custom Tools"
family = "gemini-pro"
release_date = "2025-05-06"
last_updated = "2025-05-06"
release_date = "2026-02-19"
last_updated = "2026-02-19"
attachment = true
reasoning = true
temperature = true
knowledge = "2025-01"
tool_call = true
structured_output = true
knowledge = "2025-01"
open_weights = false
[cost]
input = 1.25
output = 10.00
cache_read = 0.31
input = 2
output = 12
cache_read = 0.2
[[cost.tiers]]
tier = { size = 200_000 }
input = 4
output = 18
cache_read = 0.4
[limit]
context = 1_048_576
@@ -15,6 +15,12 @@ input = 2
output = 12
cache_read = 0.2
[[cost.tiers]]
tier = { size = 200_000 }
input = 4.00
output = 18.00
cache_read = 0.40
[limit]
context = 1_048_576
output = 65_536
-25
View File
@@ -1,25 +0,0 @@
name = "GLM-5"
family = "glm"
release_date = "2026-02-11"
last_updated = "2026-02-11"
attachment = false
reasoning = true
temperature = true
tool_call = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.88
output = 2.816
cache_read = 0.176
[limit]
context = 202_752
output = 0
[modalities]
input = ["text"]
output = ["text"]
@@ -1,11 +1,10 @@
name = "Qwen3.6-27B"
family = "qwen3.6"
release_date = "2026-04-02"
last_updated = "2026-04-02"
name = "GLM 5 Vision Turbo"
family = "glmv"
release_date = "2026-05-09"
last_updated = "2026-05-09"
attachment = true
reasoning = true
temperature = true
knowledge = "2025-04"
tool_call = true
structured_output = true
open_weights = false
@@ -14,12 +13,13 @@ open_weights = false
field = "reasoning_content"
[cost]
input = 0.6
output = 3.6
input = 0.7042
output = 3.09848
cache_read = 0.169008
[limit]
context = 262_144
output = 65_536
context = 200_000
output = 128_000
[modalities]
input = ["text", "image", "video"]
@@ -1,23 +0,0 @@
name = "GPT-4.1 mini"
family = "gpt-mini"
release_date = "2025-04-14"
last_updated = "2025-04-14"
attachment = true
reasoning = false
temperature = true
knowledge = "2024-04"
tool_call = true
open_weights = false
[cost]
input = 0.40
output = 1.60
cache_read = 0.10
[limit]
context = 1_047_576
output = 32_768
[modalities]
input = ["text", "image"]
output = ["text"]
-23
View File
@@ -1,23 +0,0 @@
name = "GPT-4.1"
family = "gpt"
release_date = "2025-04-14"
last_updated = "2025-04-14"
attachment = true
reasoning = false
temperature = true
knowledge = "2024-04"
tool_call = true
open_weights = false
[cost]
input = 2.00
output = 8.00
cache_read = 0.50
[limit]
context = 1_047_576
output = 32_768
[modalities]
input = ["text", "image"]
output = ["text"]
@@ -5,20 +5,20 @@ last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
knowledge = "2024-09-30"
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[cost]
input = 0.25
output = 2
output = 2.00
cache_read = 0.025
[limit]
context = 400_000
output = 128_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
+3 -3
View File
@@ -5,20 +5,20 @@ last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
knowledge = "2024-09-30"
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[cost]
input = 1.25
output = 10
output = 10.00
cache_read = 0.125
[limit]
context = 400_000
output = 128_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
+7 -5
View File
@@ -1,21 +1,23 @@
name = "GPT-5.1"
family = "gpt"
release_date = "2025-11-15"
last_updated = "2025-11-15"
release_date = "2025-11-13"
last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = true
temperature = false
knowledge = "2024-09-30"
tool_call = true
knowledge = "2025-11"
structured_output = true
open_weights = false
[cost]
input = 1.25
output = 10.00
cache_read = 0.125
cache_read = 0.13
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
+6 -4
View File
@@ -1,9 +1,10 @@
name = "GPT-5.2-Codex"
name = "GPT-5.2 Codex"
family = "gpt-codex"
release_date = "2026-01-14"
last_updated = "2026-01-14"
release_date = "2025-12-11"
last_updated = "2025-12-11"
attachment = true
reasoning = true
temperature = false
knowledge = "2025-08-31"
tool_call = true
structured_output = true
@@ -16,8 +17,9 @@ cache_read = 0.175
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
input = ["text", "image", "pdf"]
output = ["text"]
+2
View File
@@ -7,6 +7,7 @@ reasoning = true
temperature = false
knowledge = "2025-08-31"
tool_call = true
structured_output = true
open_weights = false
[cost]
@@ -16,6 +17,7 @@ cache_read = 0.175
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
+3 -3
View File
@@ -5,20 +5,20 @@ last_updated = "2026-02-05"
attachment = true
reasoning = true
temperature = false
knowledge = "2025-08-31"
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[cost]
input = 1.75
output = 14
output = 14.00
cache_read = 0.175
[limit]
context = 400_000
output = 128_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
+11 -5
View File
@@ -1,13 +1,14 @@
name = "GPT-5.4-Mini"
name = "GPT-5.4 mini"
family = "gpt-mini"
release_date = "2026-03-11"
last_updated = "2026-03-11"
release_date = "2026-03-17"
last_updated = "2026-03-17"
attachment = true
reasoning = false
reasoning = true
temperature = false
knowledge = "2025-08-31"
tool_call = true
open_weights = false
structured_output = true
open_weights = false
[cost]
input = 0.75
@@ -16,8 +17,13 @@ cache_read = 0.075
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
[experimental.modes.fast]
cost = { input = 1.50, output = 9.00, cache_read = 0.15 }
provider = { body = { service_tier = "priority" } }
+17 -5
View File
@@ -1,23 +1,35 @@
name = "GPT-5.4"
family = "gpt"
release_date = "2026-03-11"
last_updated = "2026-03-11"
release_date = "2026-03-05"
last_updated = "2026-03-05"
attachment = true
reasoning = true
temperature = false
knowledge = "2025-08-31"
tool_call = true
open_weights = false
structured_output = true
open_weights = false
[cost]
input = 2.50
output = 15.00
cache_read = 0.25
[[cost.tiers]]
tier = { size = 272_000 }
input = 5.00
output = 22.50
cache_read = 0.50
[limit]
context = 400_000
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image"]
input = ["text", "image", "pdf"]
output = ["text"]
[experimental.modes.fast]
cost = { input = 5.00, output = 30.00, cache_read = 0.50 }
provider = { body = { service_tier = "priority" } }
+15 -4
View File
@@ -5,20 +5,31 @@ last_updated = "2026-04-23"
attachment = true
reasoning = true
temperature = false
knowledge = "2025-12-01"
tool_call = true
structured_output = true
knowledge = "2025-12-01"
open_weights = false
[cost]
input = 5
output = 30
cache_read = 0.5
input = 5.00
output = 30.00
cache_read = 0.50
[[cost.tiers]]
tier = { size = 272_000 }
input = 10.00
output = 45.00
cache_read = 1.00
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[experimental.modes.fast]
cost = { input = 12.50, output = 75.00, cache_read = 1.25 }
provider = { body = { service_tier = "priority" } }
+29
View File
@@ -0,0 +1,29 @@
name = "Grok 4.3"
family = "grok"
release_date = "2026-05-01"
last_updated = "2026-05-01"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = false
[cost]
input = 1.25
output = 2.5
cache_read = 0.2
[[cost.tiers]]
tier = { size = 200_000 }
input = 2.5
output = 5.0
cache_read = 0.4
[limit]
context = 1_000_000
output = 1_000_000
[modalities]
input = ["text", "image"]
output = ["text"]
@@ -1,23 +0,0 @@
name = "Kimi-K2-Thinking"
family = "kimi"
release_date = "2025-11-06"
last_updated = "2025-11-06"
attachment = false
reasoning = true
temperature = true
knowledge = "2025-11"
tool_call = true
open_weights = true
[cost]
input = 0.55
output = 2.19
cache_read = 0.14
[limit]
context = 128_000
output = 64_000
[modalities]
input = ["text"]
output = ["text"]
+4 -4
View File
@@ -2,7 +2,7 @@ name = "Kimi K2.5"
family = "kimi-k2.5"
release_date = "2026-01"
last_updated = "2026-01"
attachment = false
attachment = true
reasoning = true
temperature = false
tool_call = true
@@ -16,11 +16,11 @@ field = "reasoning_content"
[cost]
input = 0.6
output = 3
cache_read = 0.105
cache_read = 0.10
[limit]
context = 256_000
output = 0
context = 262_144
output = 32_768
[modalities]
input = ["text", "image", "video"]
+4 -4
View File
@@ -4,7 +4,7 @@ release_date = "2026-04-21"
last_updated = "2026-04-21"
attachment = true
reasoning = true
temperature = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-01"
@@ -15,12 +15,12 @@ field = "reasoning_content"
[cost]
input = 0.95
output = 3.9995
cache_read = 0.160835
output = 4
cache_read = 0.16
[limit]
context = 262_144
output = 262_144
output = 32_768
[modalities]
input = ["text", "image", "video"]
@@ -1,24 +0,0 @@
name = "MiniMax-M2.1"
family = "minimax"
release_date = "2025-12-23"
last_updated = "2025-12-23"
attachment = false
reasoning = true
temperature = true
tool_call = true
open_weights = true
[interleaved]
field = "reasoning_details"
[cost]
input = 0.288
output = 1.152
[limit]
context = 204_800
output = 192_000
[modalities]
input = ["text"]
output = ["text"]
+10 -5
View File
@@ -1,4 +1,4 @@
name = "MiniMax-M2.7"
name = "MiniMax M2.7"
family = "minimax"
release_date = "2026-03-18"
last_updated = "2026-03-18"
@@ -6,15 +6,20 @@ attachment = false
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = true
[interleaved]
field = "reasoning_content"
[cost]
input = 0.2958
output = 1.1832
cache_read = 0.05916
input = 0.3
output = 1.2
cache_read = 0.06
cache_write = 0.375
[limit]
context = 200_000
context = 204_800
output = 128_000
[modalities]
@@ -1,24 +0,0 @@
name = "Qwen3 Coder Plus"
family = "qwen"
release_date = "2025-07-23"
last_updated = "2025-07-23"
attachment = false
reasoning = false
temperature = true
tool_call = true
knowledge = "2025-04"
open_weights = true
[cost]
input = 0.137
output = 0.548
cache_read = 0.137
[limit]
context = 2_000_000
output = 64_000
input = 262_144
[modalities]
input = ["text"]
output = ["text"]
@@ -1,23 +0,0 @@
name = "Qwen3 Max"
family = "qwen"
release_date = "2025-09-23"
last_updated = "2025-09-23"
attachment = false
reasoning = false
temperature = true
tool_call = true
knowledge = "2025-04"
open_weights = false
[cost]
input = 0.34246
output = 1.36984
cache_read = 0.34246
[limit]
context = 252_000
output = 32_000
[modalities]
input = ["text"]
output = ["text"]
@@ -1,24 +0,0 @@
name = "Qwen3.5 Plus"
family = "qwen"
release_date = "2026-02-16"
last_updated = "2026-02-16"
attachment = false
reasoning = true
temperature = true
tool_call = true
knowledge = "2025-04"
open_weights = false
[cost]
input = 0.1096
output = 0.6576
cache_read = 0.01096
cache_write = 0.137
[limit]
context = 991_000
output = 64_000
[modalities]
input = ["text", "image", "video"]
output = ["text"]
+18 -7
View File
@@ -1,23 +1,34 @@
name = "Qwen3.6 Plus"
family = "qwen"
name = "Qwen3.6 Flash"
family = "qwen3.6"
release_date = "2026-04-02"
last_updated = "2026-04-02"
attachment = false
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-04"
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.169
output = 1.014
input = 0.17
output = 1.01
cache_read = 0.0169
cache_write = 0.21125
[[cost.tiers]]
tier = { size = 256_000 }
input = 0.68
output = 4.06
cache_read = 0.0676
cache_write = 0.845
[limit]
context = 1_000_000
output = 65_536
context = 991_000
output = 64_000
[modalities]
input = ["text", "image", "video"]
@@ -0,0 +1,35 @@
name = "Qwen3.6 Max Preview"
family = "qwen3.6"
release_date = "2026-05-09"
last_updated = "2026-05-09"
attachment = false
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-04"
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 1.27
output = 7.61
cache_read = 0.1268
cache_write = 1.585
[[cost.tiers]]
tier = { size = 128_000 }
input = 2.11
output = 12.67
cache_read = 0.2112
cache_write = 2.64
[limit]
context = 240_000
output = 64_000
[modalities]
input = ["text"]
output = ["text"]
@@ -0,0 +1,35 @@
name = "Qwen3.6 Plus"
family = "qwen3.6"
release_date = "2026-05-09"
last_updated = "2026-05-09"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-04"
open_weights = false
[interleaved]
field = "reasoning_content"
[cost]
input = 0.28
output = 1.69
cache_read = 0.0282
cache_write = 0.3525
[[cost.tiers]]
tier = { size = 256_000 }
input = 1.13
output = 6.77
cache_read = 0.1128
cache_write = 1.41
[limit]
context = 991_000
output = 64_000
[modalities]
input = ["text", "image", "video"]
output = ["text"]
@@ -0,0 +1,16 @@
name = "Xiaomi MiMo-V2.5 (free)"
family = "mimo-v2.5"
last_updated = "2026-05-13"
[extends]
from = "xiaomi/mimo-v2.5"
omit = ["cost.context_over_200k", "cost.tiers"]
[cost]
input = 0
output = 0
cache_read = 0
[limit]
context = 1_048_576
output = 131_072
@@ -0,0 +1,16 @@
name = "Xiaomi MiMo-V2.5-Pro (free)"
family = "mimo-v2.5-pro"
last_updated = "2026-05-13"
[extends]
from = "xiaomi/mimo-v2.5-pro"
omit = ["cost.context_over_200k", "cost.tiers"]
[cost]
input = 0
output = 0
cache_read = 0
[limit]
context = 1_048_576
output = 131_072
@@ -0,0 +1,17 @@
name = "Xiaomi MiMo-V2.5-Pro"
family = "mimo-v2.5-pro"
last_updated = "2026-05-13"
[extends]
from = "xiaomi/mimo-v2.5-pro"
[cost]
input = 1.10
output = 3.30
cache_read = 0.22
[[cost.tiers]]
tier = { size = 256_000 }
input = 2.20
output = 6.60
cache_read = 0.44
@@ -0,0 +1,17 @@
name = "Xiaomi MiMo-V2.5"
family = "mimo-v2.5"
last_updated = "2026-05-13"
[extends]
from = "xiaomi/mimo-v2.5"
[cost]
input = 0.44
output = 2.20
cache_read = 0.088
[[cost.tiers]]
tier = { size = 256_000 }
input = 0.88
output = 4.40
cache_read = 0.176
@@ -1,4 +1,4 @@
name = "GLM-5.1"
name = "GLM-5.1 (Z.ai)"
family = "glm"
release_date = "2026-03-27"
last_updated = "2026-03-27"
@@ -7,7 +7,7 @@ reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = false
open_weights = true
[interleaved]
field = "reasoning_content"
+11 -4
View File
@@ -10,10 +10,17 @@ tool_call = true
open_weights = false
[cost]
input = 0.276
output = 1.651
cache_read = 0.028
cache_write = 0.344
input = 0.50
output = 3.00
cache_read = 0.05
cache_write = 0.625
[[cost.tiers]]
tier = { size = 256_000 }
input = 2.00
output = 6.00
cache_read = 0.20
cache_write = 2.50
[limit]
context = 1_000_000
+11 -4
View File
@@ -10,10 +10,17 @@ tool_call = true
open_weights = false
[cost]
input = 0.276
output = 1.651
cache_read = 0.028
cache_write = 0.344
input = 0.50
output = 3.00
cache_read = 0.05
cache_write = 0.625
[[cost.tiers]]
tier = { size = 256_000 }
input = 2.00
output = 6.00
cache_read = 0.20
cache_write = 2.50
[limit]
context = 1_000_000
+23
View File
@@ -0,0 +1,23 @@
name = "Qwen3.7 Max"
family = "qwen"
release_date = "2026-05-21"
last_updated = "2026-05-21"
attachment = false
reasoning = true
temperature = true
tool_call = true
open_weights = false
[cost]
input = 2.50
output = 7.50
cache_read = 0.50
cache_write = 3.125
[limit]
context = 1_000_000
output = 65_536
[modalities]
input = ["text"]
output = ["text"]
@@ -1,2 +0,0 @@
[extends]
from = "anthropic/claude-3-5-haiku-20241022"
@@ -1,2 +0,0 @@
[extends]
from = "anthropic/claude-3-5-sonnet-20240620"
@@ -1,24 +0,0 @@
name = "Claude Sonnet 3.5 v2"
family = "claude-sonnet"
release_date = "2024-10-22"
last_updated = "2024-10-22"
attachment = true
reasoning = false
temperature = true
knowledge = "2024-04"
tool_call = true
open_weights = false
[cost]
input = 3.00
output = 15.00
cache_read = 0.30
cache_write = 3.75
[limit]
context = 200_000
output = 8_192
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
@@ -1,24 +0,0 @@
name = "Claude Sonnet 3.7"
family = "claude-sonnet"
release_date = "2025-02-19"
last_updated = "2025-02-19"
attachment = true
reasoning = false
temperature = true
knowledge = "2024-04"
tool_call = true
open_weights = false
[cost]
input = 3.00
output = 15.00
cache_read = 0.30
cache_write = 3.75
[limit]
context = 200_000
output = 8_192
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
@@ -1,2 +0,0 @@
[extends]
from = "anthropic/claude-opus-4-20250514"
@@ -6,7 +6,6 @@ attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-01-31"
open_weights = false
@@ -1,2 +0,0 @@
[extends]
from = "anthropic/claude-sonnet-4-20250514"
@@ -1,2 +1,4 @@
structured_output = true
[extends]
from = "anthropic/claude-sonnet-4-6"
@@ -0,0 +1,5 @@
name = "Claude Haiku 4.5 (AU)"
structured_output = true
[extends]
from = "anthropic/claude-haiku-4-5-20251001"
@@ -0,0 +1,5 @@
name = "Claude Sonnet 4.5 (AU)"
structured_output = true
[extends]
from = "anthropic/claude-sonnet-4-5"
@@ -7,6 +7,7 @@ reasoning = true
temperature = true
knowledge = "2024-07"
tool_call = true
structured_output = true
open_weights = true
[cost]
@@ -6,7 +6,6 @@ attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-01-31"
open_weights = false
@@ -1,4 +0,0 @@
name = "Claude Sonnet 4 (EU)"
[extends]
from = "anthropic/claude-sonnet-4-20250514"
@@ -1,4 +1,5 @@
name = "Claude Sonnet 4.6 (EU)"
structured_output = true
[extends]
from = "anthropic/claude-sonnet-4-6"
@@ -6,7 +6,6 @@ attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-01-31"
open_weights = false
@@ -1,4 +0,0 @@
name = "Claude Sonnet 4 (Global)"
[extends]
from = "anthropic/claude-sonnet-4-20250514"

Some files were not shown because too many files have changed in this diff Show More