From c2c5cc8f21d590edb2f52219d20016606c0d00df Mon Sep 17 00:00:00 2001 From: Aiden Cline Date: Fri, 15 May 2026 10:00:32 -0500 Subject: [PATCH] add sync script for openrouter, sync openrouter models --- models.json | 1 + package.json | 3 +- packages/core/script/sync-openrouter.ts | 309 ++++++++++++++++++ .../models/ai21/jamba-large-1.7.toml | 23 ++ .../models/aion-labs/aion-1.0-mini.toml | 22 ++ .../openrouter/models/aion-labs/aion-1.0.toml | 22 ++ .../openrouter/models/aion-labs/aion-2.0.toml | 23 ++ .../aion-labs/aion-rp-llama-3.1-8b.toml | 23 ++ .../codellama-7b-instruct-solidity.toml | 23 ++ .../alibaba/tongyi-deepresearch-30b-a3b.toml | 24 ++ .../models/allenai/olmo-3-32b-think.toml | 22 ++ .../models/amazon/nova-2-lite-v1.toml | 22 ++ .../models/amazon/nova-lite-v1.toml | 23 ++ .../models/amazon/nova-micro-v1.toml | 23 ++ .../models/amazon/nova-premier-v1.toml | 23 ++ .../openrouter/models/amazon/nova-pro-v1.toml | 23 ++ .../models/anthracite-org/magnum-v4-72b.toml | 23 ++ .../models/anthropic/claude-3-haiku.toml | 25 ++ .../models/anthropic/claude-3.5-haiku.toml | 27 +- .../models/anthropic/claude-haiku-4.5.toml | 6 +- .../models/anthropic/claude-opus-4.1.toml | 10 +- .../models/anthropic/claude-opus-4.5.toml | 10 +- .../anthropic/claude-opus-4.6-fast.toml | 24 ++ .../models/anthropic/claude-opus-4.6.toml | 18 +- .../anthropic/claude-opus-4.7-fast.toml | 24 ++ .../models/anthropic/claude-opus-4.7.toml | 14 +- .../models/anthropic/claude-opus-4.toml | 27 +- .../models/anthropic/claude-sonnet-4.5.toml | 16 +- .../models/anthropic/claude-sonnet-4.6.toml | 16 +- .../models/anthropic/claude-sonnet-4.toml | 21 +- .../models/arcee-ai/coder-large.toml | 23 ++ .../models/arcee-ai/maestro-reasoning.toml | 23 ++ .../openrouter/models/arcee-ai/spotlight.toml | 23 ++ ...w:free.toml => trinity-large-preview.toml} | 14 +- .../arcee-ai/trinity-large-thinking.toml | 10 +- .../arcee-ai/trinity-large-thinking:free.toml | 22 ++ .../trinity-mini.toml} | 14 +- .../models/arcee-ai/virtuoso-large.toml | 23 ++ .../openrouter/models/baidu/cobuddy:free.toml | 22 ++ .../baidu/ernie-4.5-21b-a3b-thinking.toml | 23 ++ .../models/baidu/ernie-4.5-21b-a3b.toml | 23 ++ .../models/baidu/ernie-4.5-300b-a47b.toml | 23 ++ .../models/baidu/ernie-4.5-vl-28b-a3b.toml | 23 ++ .../models/baidu/ernie-4.5-vl-424b-a47b.toml | 23 ++ .../models/baidu/qianfan-ocr-fast.toml | 22 ++ .../models/black-forest-labs/flux.2-flex.toml | 23 -- .../black-forest-labs/flux.2-klein-4b.toml | 23 -- .../models/black-forest-labs/flux.2-max.toml | 23 -- .../models/black-forest-labs/flux.2-pro.toml | 23 -- .../models/bytedance-seed/seed-1.6-flash.toml | 22 ++ .../models/bytedance-seed/seed-1.6.toml | 22 ++ .../models/bytedance-seed/seed-2.0-lite.toml | 22 ++ .../models/bytedance-seed/seed-2.0-mini.toml | 22 ++ .../models/bytedance-seed/seedream-4.5.toml | 23 -- .../models/bytedance/ui-tars-1.5-7b.toml | 23 ++ ...lphin-mistral-24b-venice-edition:free.toml | 9 +- .../command-a.toml} | 17 +- .../models/cohere/command-r-08-2024.toml | 23 ++ .../models/cohere/command-r-plus-08-2024.toml | 23 ++ .../models/cohere/command-r7b-12-2024.toml | 23 ++ .../models/deepcogito/cogito-v2.1-671b.toml | 22 ++ .../deepseek/deepseek-chat-v3-0324.toml | 18 +- .../models/deepseek/deepseek-chat-v3.1.toml | 11 +- .../models/deepseek/deepseek-chat.toml | 23 ++ .../models/deepseek/deepseek-r1-0528.toml | 24 ++ .../deepseek-r1-distill-llama-70b.toml | 17 +- .../deepseek-r1-distill-qwen-32b.toml | 23 ++ .../models/deepseek/deepseek-r1.toml | 8 +- .../deepseek/deepseek-v3.1-terminus.toml | 9 +- ...nus:exacto.toml => deepseek-v3.2-exp.toml} | 12 +- .../deepseek/deepseek-v3.2-speciale.toml | 11 +- .../models/deepseek/deepseek-v3.2.toml | 9 +- .../models/deepseek/deepseek-v4-flash.toml | 23 +- .../deepseek/deepseek-v4-flash:free.toml | 22 ++ .../models/deepseek/deepseek-v4-pro.toml | 23 +- .../models/essentialai/rnj-1-instruct.toml | 22 ++ .../models/google/gemini-2.0-flash-001.toml | 13 +- .../google/gemini-2.0-flash-lite-001.toml | 23 ++ .../models/google/gemini-2.5-flash-image.toml | 25 ++ ...gemini-2.5-flash-lite-preview-09-2025.toml | 16 +- .../models/google/gemini-2.5-flash-lite.toml | 20 +- .../models/google/gemini-2.5-flash.toml | 20 +- .../google/gemini-2.5-pro-preview-05-06.toml | 16 +- ...06-05.toml => gemini-2.5-pro-preview.toml} | 12 +- .../models/google/gemini-2.5-pro.toml | 28 +- .../models/google/gemini-3-flash-preview.toml | 10 +- .../google/gemini-3-pro-image-preview.toml | 25 ++ .../models/google/gemini-3-pro-preview.toml | 26 -- .../gemini-3.1-flash-image-preview.toml | 12 +- .../google/gemini-3.1-flash-lite-preview.toml | 8 +- .../models/google/gemini-3.1-flash-lite.toml | 25 ++ .../gemini-3.1-pro-preview-customtools.toml | 22 +- .../models/google/gemini-3.1-pro-preview.toml | 18 +- ...gemma-2-9b-it.toml => gemma-2-27b-it.toml} | 15 +- .../models/google/gemma-3-12b-it.toml | 10 +- .../models/google/gemma-3-27b-it.toml | 10 +- .../models/google/gemma-3-27b-it:free.toml | 22 -- .../models/google/gemma-3-4b-it.toml | 11 +- .../models/google/gemma-3-4b-it:free.toml | 22 -- .../models/google/gemma-3n-e2b-it:free.toml | 22 -- .../models/google/gemma-3n-e4b-it.toml | 11 +- .../models/google/gemma-3n-e4b-it:free.toml | 22 -- .../models/google/gemma-4-26b-a4b-it.toml | 11 +- .../google/gemma-4-26b-a4b-it:free.toml | 6 +- .../models/google/gemma-4-31b-it.toml | 10 +- .../models/google/gemma-4-31b-it:free.toml | 4 +- .../models/google/lyria-3-clip-preview.toml | 22 ++ .../models/google/lyria-3-pro-preview.toml | 22 ++ .../models/gryphe/mythomax-l2-13b.toml | 23 ++ .../ibm-granite/granite-4.0-h-micro.toml | 22 ++ .../models/ibm-granite/granite-4.1-8b.toml | 23 ++ .../models/inception/mercury-edit-2.toml | 21 -- .../models/inclusionai/ling-2.6-1t.toml | 23 ++ .../models/inclusionai/ling-2.6-flash.toml | 23 ++ .../models/inclusionai/ring-2.6-1t:free.toml | 22 ++ .../models/inflection/inflection-3-pi.toml | 23 ++ .../inflection/inflection-3-productivity.toml | 23 ++ .../models/kwaipilot/kat-coder-pro-v2.toml | 23 ++ .../models/liquid/lfm-2-24b-a2b.toml | 22 ++ .../liquid/lfm-2.5-1.2b-instruct:free.toml | 12 +- .../liquid/lfm-2.5-1.2b-thinking:free.toml | 12 +- .../openrouter/models/mancer/weaver.toml | 23 ++ .../meta-llama/llama-3-70b-instruct.toml | 23 ++ .../meta-llama/llama-3-8b-instruct.toml | 23 ++ .../meta-llama/llama-3.1-70b-instruct.toml | 23 ++ .../meta-llama/llama-3.1-8b-instruct.toml | 23 ++ .../llama-3.2-11b-vision-instruct.toml | 16 +- .../meta-llama/llama-3.2-1b-instruct.toml | 23 ++ .../meta-llama/llama-3.2-3b-instruct.toml | 23 ++ .../llama-3.2-3b-instruct:free.toml | 19 +- .../meta-llama/llama-3.3-70b-instruct.toml | 23 ++ .../llama-3.3-70b-instruct:free.toml | 12 +- .../models/meta-llama/llama-4-maverick.toml | 23 ++ .../models/meta-llama/llama-4-scout.toml | 23 ++ .../models/meta-llama/llama-guard-3-8b.toml | 23 ++ .../models/meta-llama/llama-guard-4-12b.toml | 23 ++ .../models/microsoft/phi-4-mini-instruct.toml | 23 ++ .../openrouter/models/microsoft/phi-4.toml | 23 ++ .../models/microsoft/wizardlm-2-8x22b.toml | 23 ++ .../openrouter/models/minimax/minimax-01.toml | 20 +- .../openrouter/models/minimax/minimax-m1.toml | 18 +- .../models/minimax/minimax-m2-her.toml | 23 ++ .../models/minimax/minimax-m2.1.toml | 20 +- .../models/minimax/minimax-m2.5.toml | 19 +- .../models/minimax/minimax-m2.5:free.toml | 18 +- .../models/minimax/minimax-m2.7.toml | 9 +- .../openrouter/models/minimax/minimax-m2.toml | 21 +- .../models/mistralai/codestral-2508.toml | 17 +- .../models/mistralai/devstral-2512.toml | 15 +- ...-medium-2507.toml => devstral-medium.toml} | 17 +- .../models/mistralai/devstral-small-2505.toml | 22 -- ...al-small-2507.toml => devstral-small.toml} | 13 +- .../models/mistralai/ministral-14b-2512.toml | 23 ++ .../models/mistralai/ministral-3b-2512.toml | 23 ++ .../models/mistralai/ministral-8b-2512.toml | 23 ++ .../mistralai/mistral-7b-instruct-v0.1.toml | 23 ++ .../models/mistralai/mistral-large-2407.toml | 24 ++ .../models/mistralai/mistral-large-2411.toml | 24 ++ .../models/mistralai/mistral-large-2512.toml | 23 ++ .../models/mistralai/mistral-large.toml | 24 ++ .../models/mistralai/mistral-medium-3-5.toml | 22 ++ .../models/mistralai/mistral-medium-3.1.toml | 15 +- .../models/mistralai/mistral-medium-3.toml | 9 +- .../models/mistralai/mistral-nemo.toml | 23 ++ .../models/mistralai/mistral-saba.toml | 24 ++ .../mistral-small-24b-instruct-2501.toml | 23 ++ .../models/mistralai/mistral-small-2603.toml | 7 +- .../mistral-small-3.1-24b-instruct.toml | 21 +- .../mistral-small-3.2-24b-instruct.toml | 19 +- .../mistralai/mixtral-8x22b-instruct.toml | 24 ++ .../models/mistralai/pixtral-large-2411.toml | 24 ++ .../mistralai/voxtral-small-24b-2507.toml | 23 ++ .../models/moonshotai/kimi-k2-0905.toml | 12 +- .../models/moonshotai/kimi-k2.5.toml | 8 +- .../models/moonshotai/kimi-k2.6.toml | 10 +- .../openrouter/models/moonshotai/kimi-k2.toml | 9 +- .../models/morph/morph-v3-fast.toml | 22 ++ .../models/morph/morph-v3-large.toml | 22 ++ .../models/nex-agi/deepseek-v3.1-nex-n1.toml | 22 ++ .../nousresearch/hermes-2-pro-llama-3-8b.toml | 23 ++ .../nousresearch/hermes-3-llama-3.1-405b.toml | 23 ++ .../hermes-3-llama-3.1-405b:free.toml | 11 +- .../nousresearch/hermes-3-llama-3.1-70b.toml | 23 ++ .../models/nousresearch/hermes-4-405b.toml | 16 +- .../models/nousresearch/hermes-4-70b.toml | 13 +- .../llama-3.3-nemotron-super-49b-v1.5.toml | 23 ++ .../nvidia/nemotron-3-nano-30b-a3b.toml | 22 ++ .../nvidia/nemotron-3-nano-30b-a3b:free.toml | 11 +- ...on-3-nano-omni-30b-a3b-reasoning:free.toml | 11 +- .../nvidia/nemotron-3-super-120b-a12b.toml | 21 +- .../nemotron-3-super-120b-a12b:free.toml | 19 +- .../nvidia/nemotron-nano-12b-v2-vl:free.toml | 14 +- .../models/nvidia/nemotron-nano-9b-v2.toml | 12 +- .../nvidia/nemotron-nano-9b-v2:free.toml | 9 +- .../models/openai/gpt-3.5-turbo-0613.toml | 23 ++ .../models/openai/gpt-3.5-turbo-16k.toml | 23 ++ .../models/openai/gpt-3.5-turbo-instruct.toml | 23 ++ .../models/openai/gpt-3.5-turbo.toml | 23 ++ .../openrouter/models/openai/gpt-4-0314.toml | 23 ++ .../models/openai/gpt-4-1106-preview.toml | 23 ++ .../models/openai/gpt-4-turbo-preview.toml | 23 ++ .../openrouter/models/openai/gpt-4-turbo.toml | 23 ++ .../models/openai/gpt-4.1-mini.toml | 10 +- .../models/openai/gpt-4.1-nano.toml | 24 ++ .../openrouter/models/openai/gpt-4.1.toml | 10 +- providers/openrouter/models/openai/gpt-4.toml | 23 ++ .../models/openai/gpt-4o-2024-05-13.toml | 23 ++ .../models/openai/gpt-4o-2024-08-06.toml | 24 ++ .../models/openai/gpt-4o-2024-11-20.toml | 24 ++ .../models/openai/gpt-4o-audio-preview.toml | 23 ++ .../models/openai/gpt-4o-mini-2024-07-18.toml | 24 ++ .../openai/gpt-4o-mini-search-preview.toml | 23 ++ .../openrouter/models/openai/gpt-4o-mini.toml | 8 +- .../models/openai/gpt-4o-search-preview.toml | 23 ++ .../openrouter/models/openai/gpt-4o.toml | 23 ++ .../openrouter/models/openai/gpt-5-chat.toml | 17 +- .../openrouter/models/openai/gpt-5-codex.toml | 10 +- .../models/openai/gpt-5-image-mini.toml | 23 ++ .../openrouter/models/openai/gpt-5-image.toml | 12 +- .../openrouter/models/openai/gpt-5-mini.toml | 9 +- .../openrouter/models/openai/gpt-5-nano.toml | 9 +- .../openrouter/models/openai/gpt-5-pro.toml | 10 +- .../models/openai/gpt-5.1-chat.toml | 10 +- .../models/openai/gpt-5.1-codex-max.toml | 14 +- .../models/openai/gpt-5.1-codex-mini.toml | 12 +- .../models/openai/gpt-5.1-codex.toml | 6 +- .../openrouter/models/openai/gpt-5.1.toml | 10 +- .../models/openai/gpt-5.2-chat.toml | 14 +- .../models/openai/gpt-5.2-codex.toml | 6 +- .../openrouter/models/openai/gpt-5.2-pro.toml | 12 +- .../openrouter/models/openai/gpt-5.2.toml | 10 +- .../models/openai/gpt-5.3-chat.toml | 23 ++ .../models/openai/gpt-5.3-codex.toml | 4 +- .../models/openai/gpt-5.4-image-2.toml | 23 ++ .../models/openai/gpt-5.4-mini.toml | 6 +- .../models/openai/gpt-5.4-nano.toml | 10 +- .../openrouter/models/openai/gpt-5.4-pro.toml | 9 +- .../openrouter/models/openai/gpt-5.4.toml | 26 +- .../openrouter/models/openai/gpt-5.5-pro.toml | 25 +- .../openrouter/models/openai/gpt-5.5.toml | 27 +- providers/openrouter/models/openai/gpt-5.toml | 9 +- .../models/openai/gpt-audio-mini.toml | 22 ++ .../openrouter/models/openai/gpt-audio.toml | 22 ++ .../models/openai/gpt-chat-latest.toml | 23 ++ .../models/openai/gpt-oss-120b.toml | 7 +- .../models/openai/gpt-oss-120b:free.toml | 8 +- .../openrouter/models/openai/gpt-oss-20b.toml | 9 +- .../models/openai/gpt-oss-20b:free.toml | 10 +- .../models/openai/gpt-oss-safeguard-20b.toml | 8 +- .../openrouter/models/openai/o1-pro.toml | 23 ++ providers/openrouter/models/openai/o1.toml | 24 ++ .../models/openai/o3-deep-research.toml | 23 ++ .../models/openai/o3-mini-high.toml | 24 ++ .../openrouter/models/openai/o3-mini.toml | 24 ++ .../openrouter/models/openai/o3-pro.toml | 23 ++ providers/openrouter/models/openai/o3.toml | 24 ++ .../models/openai/o4-mini-deep-research.toml | 23 ++ .../models/openai/o4-mini-high.toml | 24 ++ .../openrouter/models/openai/o4-mini.toml | 12 +- .../openrouter/models/openrouter/auto.toml | 18 + .../models/openrouter/bodybuilder.toml | 18 + .../openrouter/models/openrouter/free.toml | 5 +- .../models/openrouter/owl-alpha.toml | 9 +- .../models/openrouter/pareto-code.toml | 13 +- .../models/perceptron/perceptron-mk1.toml | 22 ++ .../perplexity/sonar-deep-research.toml | 23 ++ .../models/perplexity/sonar-pro-search.toml | 22 ++ .../models/perplexity/sonar-pro.toml | 22 ++ .../perplexity/sonar-reasoning-pro.toml | 22 ++ .../openrouter/models/perplexity/sonar.toml | 22 ++ .../models/poolside/laguna-m.1:free.toml | 12 +- .../models/poolside/laguna-xs.2:free.toml | 12 +- .../models/prime-intellect/intellect-3.toml | 19 +- .../models/qwen/qwen-2.5-72b-instruct.toml | 23 ++ .../models/qwen/qwen-2.5-7b-instruct.toml | 23 ++ .../qwen/qwen-2.5-coder-32b-instruct.toml | 17 +- .../models/qwen/qwen-plus-2025-07-28.toml | 24 ++ .../qwen/qwen-plus-2025-07-28:thinking.toml | 24 ++ .../openrouter/models/qwen/qwen-plus.toml | 10 +- .../models/qwen/qwen2.5-vl-72b-instruct.toml | 15 +- .../openrouter/models/qwen/qwen3-14b.toml | 23 ++ ...b-07-25.toml => qwen3-235b-a22b-2507.toml} | 12 +- .../qwen/qwen3-235b-a22b-thinking-2507.toml | 8 +- .../models/qwen/qwen3-235b-a22b.toml | 23 ++ .../qwen/qwen3-30b-a3b-instruct-2507.toml | 10 +- .../qwen/qwen3-30b-a3b-thinking-2507.toml | 15 +- .../openrouter/models/qwen/qwen3-30b-a3b.toml | 23 ++ .../openrouter/models/qwen/qwen3-32b.toml | 23 ++ .../openrouter/models/qwen/qwen3-8b.toml | 24 ++ .../qwen/qwen3-coder-30b-a3b-instruct.toml | 4 +- .../models/qwen/qwen3-coder-flash.toml | 16 +- .../qwen3-coder-next.toml} | 16 +- .../models/qwen/qwen3-coder-plus.toml | 4 +- .../openrouter/models/qwen/qwen3-coder.toml | 12 +- ...oder:exacto.toml => qwen3-coder:free.toml} | 14 +- .../qwen3-max-thinking.toml} | 12 +- .../openrouter/models/qwen/qwen3-max.toml | 15 +- .../qwen/qwen3-next-80b-a3b-instruct.toml | 8 +- .../qwen3-next-80b-a3b-instruct:free.toml | 23 ++ .../qwen/qwen3-next-80b-a3b-thinking.toml | 10 +- .../qwen/qwen3-vl-235b-a22b-instruct.toml | 24 ++ .../qwen/qwen3-vl-235b-a22b-thinking.toml | 23 ++ .../qwen/qwen3-vl-30b-a3b-instruct.toml | 23 ++ .../qwen/qwen3-vl-30b-a3b-thinking.toml | 23 ++ .../models/qwen/qwen3-vl-32b-instruct.toml | 22 ++ .../models/qwen/qwen3-vl-8b-instruct.toml | 22 ++ .../models/qwen/qwen3-vl-8b-thinking.toml | 22 ++ .../models/qwen/qwen3.5-122b-a10b.toml | 22 ++ .../openrouter/models/qwen/qwen3.5-27b.toml | 22 ++ .../models/qwen/qwen3.5-35b-a3b.toml | 23 ++ .../models/qwen/qwen3.5-397b-a17b.toml | 7 +- .../openrouter/models/qwen/qwen3.5-9b.toml | 22 ++ .../models/qwen/qwen3.5-flash-02-23.toml | 3 +- .../models/qwen/qwen3.5-plus-02-15.toml | 7 +- .../models/qwen/qwen3.5-plus-20260420.toml | 22 ++ .../{qwen-3.6-27b.toml => qwen3.6-27b.toml} | 14 +- .../models/qwen/qwen3.6-35b-a3b.toml | 23 ++ .../openrouter/models/qwen/qwen3.6-flash.toml | 23 ++ .../models/qwen/qwen3.6-max-preview.toml | 23 ++ .../openrouter/models/qwen/qwen3.6-plus.toml | 3 +- .../openrouter/models/rekaai/reka-edge.toml | 22 ++ .../models/rekaai/reka-flash-3.toml | 23 ++ .../models/relace/relace-apply-3.toml | 21 ++ .../models/relace/relace-search.toml | 21 ++ .../models/sao10k/l3-euryale-70b.toml | 23 ++ .../models/sao10k/l3-lunaris-8b.toml | 23 ++ .../models/sao10k/l3.1-70b-hanami-x1.toml | 23 ++ .../models/sao10k/l3.1-euryale-70b.toml | 23 ++ .../models/sao10k/l3.3-euryale-70b.toml | 23 ++ .../sourceful/riverflow-v2-fast-preview.toml | 23 -- .../sourceful/riverflow-v2-max-preview.toml | 23 -- .../riverflow-v2-standard-preview.toml | 23 -- .../models/stepfun/step-3.5-flash.toml | 10 +- .../openrouter/models/switchpoint/router.toml | 22 ++ .../models/tencent/hunyuan-a13b-instruct.toml | 23 ++ .../models/tencent/hy3-preview.toml | 12 +- .../models/thedrummer/cydonia-24b-v4.1.toml | 24 ++ .../models/thedrummer/rocinante-12b.toml | 23 ++ .../models/thedrummer/skyfall-36b-v2.toml | 23 ++ .../models/thedrummer/unslopnemo-12b.toml | 23 ++ .../models/undi95/remm-slerp-l2-13b.toml | 22 ++ .../models/upstage/solar-pro-3.toml | 23 ++ .../openrouter/models/writer/palmyra-x5.toml | 22 ++ .../openrouter/models/x-ai/grok-3-beta.toml | 12 +- .../models/x-ai/grok-3-mini-beta.toml | 12 +- .../openrouter/models/x-ai/grok-3-mini.toml | 11 +- providers/openrouter/models/x-ai/grok-3.toml | 11 +- .../openrouter/models/x-ai/grok-4-fast.toml | 15 +- .../openrouter/models/x-ai/grok-4.1-fast.toml | 13 +- .../models/x-ai/grok-4.20-beta.toml | 29 -- .../x-ai/grok-4.20-multi-agent-beta.toml | 29 -- .../models/x-ai/grok-4.20-multi-agent.toml | 24 ++ .../openrouter/models/x-ai/grok-4.20.toml | 24 ++ .../openrouter/models/x-ai/grok-4.3.toml | 26 +- providers/openrouter/models/x-ai/grok-4.toml | 15 +- .../models/x-ai/grok-code-fast-1.toml | 6 +- .../models/xiaomi/mimo-v2-flash.toml | 27 +- .../models/xiaomi/mimo-v2-omni.toml | 27 +- .../openrouter/models/xiaomi/mimo-v2-pro.toml | 27 +- .../models/xiaomi/mimo-v2.5-pro.toml | 25 +- .../openrouter/models/xiaomi/mimo-v2.5.toml | 27 +- .../openrouter/models/z-ai/glm-4-32b.toml | 23 ++ .../openrouter/models/z-ai/glm-4.5-air.toml | 17 +- .../models/z-ai/glm-4.5-air:free.toml | 15 +- providers/openrouter/models/z-ai/glm-4.5.toml | 15 +- .../openrouter/models/z-ai/glm-4.5v.toml | 10 +- providers/openrouter/models/z-ai/glm-4.6.toml | 12 +- .../models/z-ai/glm-4.6:exacto.toml | 24 -- .../openrouter/models/z-ai/glm-4.6v.toml | 23 ++ .../openrouter/models/z-ai/glm-4.7-flash.toml | 11 +- providers/openrouter/models/z-ai/glm-4.7.toml | 12 +- .../openrouter/models/z-ai/glm-5-turbo.toml | 13 +- providers/openrouter/models/z-ai/glm-5.1.toml | 8 +- providers/openrouter/models/z-ai/glm-5.toml | 12 +- .../openrouter/models/z-ai/glm-5v-turbo.toml | 23 ++ .../~anthropic/claude-haiku-latest.toml | 24 ++ .../models/~anthropic/claude-opus-latest.toml | 24 ++ .../claude-sonnet-latest.toml} | 18 +- .../gemini-flash-latest.toml} | 17 +- .../models/~google/gemini-pro-latest.toml | 25 ++ .../models/~moonshotai/kimi-latest.toml | 23 ++ .../openrouter/models/~openai/gpt-latest.toml | 24 ++ .../models/~openai/gpt-mini-latest.toml | 24 ++ 383 files changed, 6041 insertions(+), 1373 deletions(-) create mode 100644 models.json create mode 100644 packages/core/script/sync-openrouter.ts create mode 100644 providers/openrouter/models/ai21/jamba-large-1.7.toml create mode 100644 providers/openrouter/models/aion-labs/aion-1.0-mini.toml create mode 100644 providers/openrouter/models/aion-labs/aion-1.0.toml create mode 100644 providers/openrouter/models/aion-labs/aion-2.0.toml create mode 100644 providers/openrouter/models/aion-labs/aion-rp-llama-3.1-8b.toml create mode 100644 providers/openrouter/models/alfredpros/codellama-7b-instruct-solidity.toml create mode 100644 providers/openrouter/models/alibaba/tongyi-deepresearch-30b-a3b.toml create mode 100644 providers/openrouter/models/allenai/olmo-3-32b-think.toml create mode 100644 providers/openrouter/models/amazon/nova-2-lite-v1.toml create mode 100644 providers/openrouter/models/amazon/nova-lite-v1.toml create mode 100644 providers/openrouter/models/amazon/nova-micro-v1.toml create mode 100644 providers/openrouter/models/amazon/nova-premier-v1.toml create mode 100644 providers/openrouter/models/amazon/nova-pro-v1.toml create mode 100644 providers/openrouter/models/anthracite-org/magnum-v4-72b.toml create mode 100644 providers/openrouter/models/anthropic/claude-3-haiku.toml create mode 100644 providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml create mode 100644 providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml create mode 100644 providers/openrouter/models/arcee-ai/coder-large.toml create mode 100644 providers/openrouter/models/arcee-ai/maestro-reasoning.toml create mode 100644 providers/openrouter/models/arcee-ai/spotlight.toml rename providers/openrouter/models/arcee-ai/{trinity-large-preview:free.toml => trinity-large-preview.toml} (59%) create mode 100644 providers/openrouter/models/arcee-ai/trinity-large-thinking:free.toml rename providers/openrouter/models/{openai/gpt-oss-120b:exacto.toml => arcee-ai/trinity-mini.toml} (57%) create mode 100644 providers/openrouter/models/arcee-ai/virtuoso-large.toml create mode 100644 providers/openrouter/models/baidu/cobuddy:free.toml create mode 100644 providers/openrouter/models/baidu/ernie-4.5-21b-a3b-thinking.toml create mode 100644 providers/openrouter/models/baidu/ernie-4.5-21b-a3b.toml create mode 100644 providers/openrouter/models/baidu/ernie-4.5-300b-a47b.toml create mode 100644 providers/openrouter/models/baidu/ernie-4.5-vl-28b-a3b.toml create mode 100644 providers/openrouter/models/baidu/ernie-4.5-vl-424b-a47b.toml create mode 100644 providers/openrouter/models/baidu/qianfan-ocr-fast.toml delete mode 100644 providers/openrouter/models/black-forest-labs/flux.2-flex.toml delete mode 100644 providers/openrouter/models/black-forest-labs/flux.2-klein-4b.toml delete mode 100644 providers/openrouter/models/black-forest-labs/flux.2-max.toml delete mode 100644 providers/openrouter/models/black-forest-labs/flux.2-pro.toml create mode 100644 providers/openrouter/models/bytedance-seed/seed-1.6-flash.toml create mode 100644 providers/openrouter/models/bytedance-seed/seed-1.6.toml create mode 100644 providers/openrouter/models/bytedance-seed/seed-2.0-lite.toml create mode 100644 providers/openrouter/models/bytedance-seed/seed-2.0-mini.toml delete mode 100644 providers/openrouter/models/bytedance-seed/seedream-4.5.toml create mode 100644 providers/openrouter/models/bytedance/ui-tars-1.5-7b.toml rename providers/openrouter/models/{google/gemma-3-12b-it:free.toml => cohere/command-a.toml} (53%) create mode 100644 providers/openrouter/models/cohere/command-r-08-2024.toml create mode 100644 providers/openrouter/models/cohere/command-r-plus-08-2024.toml create mode 100644 providers/openrouter/models/cohere/command-r7b-12-2024.toml create mode 100644 providers/openrouter/models/deepcogito/cogito-v2.1-671b.toml create mode 100644 providers/openrouter/models/deepseek/deepseek-chat.toml create mode 100644 providers/openrouter/models/deepseek/deepseek-r1-0528.toml create mode 100644 providers/openrouter/models/deepseek/deepseek-r1-distill-qwen-32b.toml rename providers/openrouter/models/deepseek/{deepseek-v3.1-terminus:exacto.toml => deepseek-v3.2-exp.toml} (60%) create mode 100644 providers/openrouter/models/deepseek/deepseek-v4-flash:free.toml create mode 100644 providers/openrouter/models/essentialai/rnj-1-instruct.toml create mode 100644 providers/openrouter/models/google/gemini-2.0-flash-lite-001.toml create mode 100644 providers/openrouter/models/google/gemini-2.5-flash-image.toml rename providers/openrouter/models/google/{gemini-2.5-pro-preview-06-05.toml => gemini-2.5-pro-preview.toml} (67%) create mode 100644 providers/openrouter/models/google/gemini-3-pro-image-preview.toml delete mode 100644 providers/openrouter/models/google/gemini-3-pro-preview.toml create mode 100644 providers/openrouter/models/google/gemini-3.1-flash-lite.toml rename providers/openrouter/models/google/{gemma-2-9b-it.toml => gemma-2-27b-it.toml} (53%) delete mode 100644 providers/openrouter/models/google/gemma-3-27b-it:free.toml delete mode 100644 providers/openrouter/models/google/gemma-3-4b-it:free.toml delete mode 100644 providers/openrouter/models/google/gemma-3n-e2b-it:free.toml delete mode 100644 providers/openrouter/models/google/gemma-3n-e4b-it:free.toml create mode 100644 providers/openrouter/models/google/lyria-3-clip-preview.toml create mode 100644 providers/openrouter/models/google/lyria-3-pro-preview.toml create mode 100644 providers/openrouter/models/gryphe/mythomax-l2-13b.toml create mode 100644 providers/openrouter/models/ibm-granite/granite-4.0-h-micro.toml create mode 100644 providers/openrouter/models/ibm-granite/granite-4.1-8b.toml delete mode 100644 providers/openrouter/models/inception/mercury-edit-2.toml create mode 100644 providers/openrouter/models/inclusionai/ling-2.6-1t.toml create mode 100644 providers/openrouter/models/inclusionai/ling-2.6-flash.toml create mode 100644 providers/openrouter/models/inclusionai/ring-2.6-1t:free.toml create mode 100644 providers/openrouter/models/inflection/inflection-3-pi.toml create mode 100644 providers/openrouter/models/inflection/inflection-3-productivity.toml create mode 100644 providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml create mode 100644 providers/openrouter/models/liquid/lfm-2-24b-a2b.toml create mode 100644 providers/openrouter/models/mancer/weaver.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3-70b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3-8b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3.1-8b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3.2-1b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3.2-3b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-3.3-70b-instruct.toml create mode 100644 providers/openrouter/models/meta-llama/llama-4-maverick.toml create mode 100644 providers/openrouter/models/meta-llama/llama-4-scout.toml create mode 100644 providers/openrouter/models/meta-llama/llama-guard-3-8b.toml create mode 100644 providers/openrouter/models/meta-llama/llama-guard-4-12b.toml create mode 100644 providers/openrouter/models/microsoft/phi-4-mini-instruct.toml create mode 100644 providers/openrouter/models/microsoft/phi-4.toml create mode 100644 providers/openrouter/models/microsoft/wizardlm-2-8x22b.toml create mode 100644 providers/openrouter/models/minimax/minimax-m2-her.toml rename providers/openrouter/models/mistralai/{devstral-medium-2507.toml => devstral-medium.toml} (58%) delete mode 100644 providers/openrouter/models/mistralai/devstral-small-2505.toml rename providers/openrouter/models/mistralai/{devstral-small-2507.toml => devstral-small.toml} (67%) create mode 100644 providers/openrouter/models/mistralai/ministral-14b-2512.toml create mode 100644 providers/openrouter/models/mistralai/ministral-3b-2512.toml create mode 100644 providers/openrouter/models/mistralai/ministral-8b-2512.toml create mode 100644 providers/openrouter/models/mistralai/mistral-7b-instruct-v0.1.toml create mode 100644 providers/openrouter/models/mistralai/mistral-large-2407.toml create mode 100644 providers/openrouter/models/mistralai/mistral-large-2411.toml create mode 100644 providers/openrouter/models/mistralai/mistral-large-2512.toml create mode 100644 providers/openrouter/models/mistralai/mistral-large.toml create mode 100644 providers/openrouter/models/mistralai/mistral-medium-3-5.toml create mode 100644 providers/openrouter/models/mistralai/mistral-nemo.toml create mode 100644 providers/openrouter/models/mistralai/mistral-saba.toml create mode 100644 providers/openrouter/models/mistralai/mistral-small-24b-instruct-2501.toml create mode 100644 providers/openrouter/models/mistralai/mixtral-8x22b-instruct.toml create mode 100644 providers/openrouter/models/mistralai/pixtral-large-2411.toml create mode 100644 providers/openrouter/models/mistralai/voxtral-small-24b-2507.toml create mode 100644 providers/openrouter/models/morph/morph-v3-fast.toml create mode 100644 providers/openrouter/models/morph/morph-v3-large.toml create mode 100644 providers/openrouter/models/nex-agi/deepseek-v3.1-nex-n1.toml create mode 100644 providers/openrouter/models/nousresearch/hermes-2-pro-llama-3-8b.toml create mode 100644 providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b.toml create mode 100644 providers/openrouter/models/nousresearch/hermes-3-llama-3.1-70b.toml create mode 100644 providers/openrouter/models/nvidia/llama-3.3-nemotron-super-49b-v1.5.toml create mode 100644 providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml create mode 100644 providers/openrouter/models/openai/gpt-3.5-turbo-0613.toml create mode 100644 providers/openrouter/models/openai/gpt-3.5-turbo-16k.toml create mode 100644 providers/openrouter/models/openai/gpt-3.5-turbo-instruct.toml create mode 100644 providers/openrouter/models/openai/gpt-3.5-turbo.toml create mode 100644 providers/openrouter/models/openai/gpt-4-0314.toml create mode 100644 providers/openrouter/models/openai/gpt-4-1106-preview.toml create mode 100644 providers/openrouter/models/openai/gpt-4-turbo-preview.toml create mode 100644 providers/openrouter/models/openai/gpt-4-turbo.toml create mode 100644 providers/openrouter/models/openai/gpt-4.1-nano.toml create mode 100644 providers/openrouter/models/openai/gpt-4.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-2024-05-13.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-2024-08-06.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-2024-11-20.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-audio-preview.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-mini-2024-07-18.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-mini-search-preview.toml create mode 100644 providers/openrouter/models/openai/gpt-4o-search-preview.toml create mode 100644 providers/openrouter/models/openai/gpt-4o.toml create mode 100644 providers/openrouter/models/openai/gpt-5-image-mini.toml create mode 100644 providers/openrouter/models/openai/gpt-5.3-chat.toml create mode 100644 providers/openrouter/models/openai/gpt-5.4-image-2.toml create mode 100644 providers/openrouter/models/openai/gpt-audio-mini.toml create mode 100644 providers/openrouter/models/openai/gpt-audio.toml create mode 100644 providers/openrouter/models/openai/gpt-chat-latest.toml create mode 100644 providers/openrouter/models/openai/o1-pro.toml create mode 100644 providers/openrouter/models/openai/o1.toml create mode 100644 providers/openrouter/models/openai/o3-deep-research.toml create mode 100644 providers/openrouter/models/openai/o3-mini-high.toml create mode 100644 providers/openrouter/models/openai/o3-mini.toml create mode 100644 providers/openrouter/models/openai/o3-pro.toml create mode 100644 providers/openrouter/models/openai/o3.toml create mode 100644 providers/openrouter/models/openai/o4-mini-deep-research.toml create mode 100644 providers/openrouter/models/openai/o4-mini-high.toml create mode 100644 providers/openrouter/models/openrouter/auto.toml create mode 100644 providers/openrouter/models/openrouter/bodybuilder.toml create mode 100644 providers/openrouter/models/perceptron/perceptron-mk1.toml create mode 100644 providers/openrouter/models/perplexity/sonar-deep-research.toml create mode 100644 providers/openrouter/models/perplexity/sonar-pro-search.toml create mode 100644 providers/openrouter/models/perplexity/sonar-pro.toml create mode 100644 providers/openrouter/models/perplexity/sonar-reasoning-pro.toml create mode 100644 providers/openrouter/models/perplexity/sonar.toml create mode 100644 providers/openrouter/models/qwen/qwen-2.5-72b-instruct.toml create mode 100644 providers/openrouter/models/qwen/qwen-2.5-7b-instruct.toml create mode 100644 providers/openrouter/models/qwen/qwen-plus-2025-07-28.toml create mode 100644 providers/openrouter/models/qwen/qwen-plus-2025-07-28:thinking.toml create mode 100644 providers/openrouter/models/qwen/qwen3-14b.toml rename providers/openrouter/models/qwen/{qwen3-235b-a22b-07-25.toml => qwen3-235b-a22b-2507.toml} (70%) create mode 100644 providers/openrouter/models/qwen/qwen3-235b-a22b.toml create mode 100644 providers/openrouter/models/qwen/qwen3-30b-a3b.toml create mode 100644 providers/openrouter/models/qwen/qwen3-32b.toml create mode 100644 providers/openrouter/models/qwen/qwen3-8b.toml rename providers/openrouter/models/{moonshotai/kimi-k2-0905:exacto.toml => qwen/qwen3-coder-next.toml} (53%) rename providers/openrouter/models/qwen/{qwen3-coder:exacto.toml => qwen3-coder:free.toml} (61%) rename providers/openrouter/models/{openrouter/elephant-alpha.toml => qwen/qwen3-max-thinking.toml} (63%) create mode 100644 providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct:free.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-235b-a22b-instruct.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-235b-a22b-thinking.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-30b-a3b-thinking.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-32b-instruct.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-8b-instruct.toml create mode 100644 providers/openrouter/models/qwen/qwen3-vl-8b-thinking.toml create mode 100644 providers/openrouter/models/qwen/qwen3.5-122b-a10b.toml create mode 100644 providers/openrouter/models/qwen/qwen3.5-27b.toml create mode 100644 providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml create mode 100644 providers/openrouter/models/qwen/qwen3.5-9b.toml create mode 100644 providers/openrouter/models/qwen/qwen3.5-plus-20260420.toml rename providers/openrouter/models/qwen/{qwen-3.6-27b.toml => qwen3.6-27b.toml} (59%) create mode 100644 providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml create mode 100644 providers/openrouter/models/qwen/qwen3.6-flash.toml create mode 100644 providers/openrouter/models/qwen/qwen3.6-max-preview.toml create mode 100644 providers/openrouter/models/rekaai/reka-edge.toml create mode 100644 providers/openrouter/models/rekaai/reka-flash-3.toml create mode 100644 providers/openrouter/models/relace/relace-apply-3.toml create mode 100644 providers/openrouter/models/relace/relace-search.toml create mode 100644 providers/openrouter/models/sao10k/l3-euryale-70b.toml create mode 100644 providers/openrouter/models/sao10k/l3-lunaris-8b.toml create mode 100644 providers/openrouter/models/sao10k/l3.1-70b-hanami-x1.toml create mode 100644 providers/openrouter/models/sao10k/l3.1-euryale-70b.toml create mode 100644 providers/openrouter/models/sao10k/l3.3-euryale-70b.toml delete mode 100644 providers/openrouter/models/sourceful/riverflow-v2-fast-preview.toml delete mode 100644 providers/openrouter/models/sourceful/riverflow-v2-max-preview.toml delete mode 100644 providers/openrouter/models/sourceful/riverflow-v2-standard-preview.toml create mode 100644 providers/openrouter/models/switchpoint/router.toml create mode 100644 providers/openrouter/models/tencent/hunyuan-a13b-instruct.toml create mode 100644 providers/openrouter/models/thedrummer/cydonia-24b-v4.1.toml create mode 100644 providers/openrouter/models/thedrummer/rocinante-12b.toml create mode 100644 providers/openrouter/models/thedrummer/skyfall-36b-v2.toml create mode 100644 providers/openrouter/models/thedrummer/unslopnemo-12b.toml create mode 100644 providers/openrouter/models/undi95/remm-slerp-l2-13b.toml create mode 100644 providers/openrouter/models/upstage/solar-pro-3.toml create mode 100644 providers/openrouter/models/writer/palmyra-x5.toml delete mode 100644 providers/openrouter/models/x-ai/grok-4.20-beta.toml delete mode 100644 providers/openrouter/models/x-ai/grok-4.20-multi-agent-beta.toml create mode 100644 providers/openrouter/models/x-ai/grok-4.20-multi-agent.toml create mode 100644 providers/openrouter/models/x-ai/grok-4.20.toml create mode 100644 providers/openrouter/models/z-ai/glm-4-32b.toml delete mode 100644 providers/openrouter/models/z-ai/glm-4.6:exacto.toml create mode 100644 providers/openrouter/models/z-ai/glm-4.6v.toml create mode 100644 providers/openrouter/models/z-ai/glm-5v-turbo.toml create mode 100644 providers/openrouter/models/~anthropic/claude-haiku-latest.toml create mode 100644 providers/openrouter/models/~anthropic/claude-opus-latest.toml rename providers/openrouter/models/{anthropic/claude-3.7-sonnet.toml => ~anthropic/claude-sonnet-latest.toml} (52%) rename providers/openrouter/models/{google/gemini-2.5-flash-preview-09-2025.toml => ~google/gemini-flash-latest.toml} (50%) create mode 100644 providers/openrouter/models/~google/gemini-pro-latest.toml create mode 100644 providers/openrouter/models/~moonshotai/kimi-latest.toml create mode 100644 providers/openrouter/models/~openai/gpt-latest.toml create mode 100644 providers/openrouter/models/~openai/gpt-mini-latest.toml diff --git a/models.json b/models.json new file mode 100644 index 000000000..3c25ba862 --- /dev/null +++ b/models.json @@ -0,0 +1 @@ +{"data":[{"id":"anthropic/claude-opus-4.7-fast","canonical_slug":"anthropic/claude-4.7-opus-fast-20260512","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.7 (Fast)","created":1778613011,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00015","web_search":"0.01","input_cache_read":"0.000003","input_cache_write":"0.0000375"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.7-opus-fast-20260512/endpoints"}},{"id":"perceptron/perceptron-mk1","canonical_slug":"perceptron/perceptron-mk1-20260512","hugging_face_id":null,"name":"Perceptron: Perceptron Mk1","created":1778597029,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/perceptron/perceptron-mk1-20260512/endpoints"}},{"id":"inclusionai/ring-2.6-1t:free","canonical_slug":"inclusionai/ring-2.6-1t-20260508","hugging_face_id":null,"name":"inclusionAI: Ring-2.6-1T (free)","created":1778247440,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/inclusionai/ring-2.6-1t-20260508/endpoints"}},{"id":"google/gemini-3.1-flash-lite","canonical_slug":"google/gemini-3.1-flash-lite-20260507","hugging_face_id":null,"name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.1-flash-lite-20260507/endpoints"}},{"id":"baidu/cobuddy:free","canonical_slug":"baidu/cobuddy-20260430","hugging_face_id":null,"name":"Baidu Qianfan: CoBuddy (free)","created":1778035480,"description":"CoBuddy is a code generation model from Baidu, optimized for coding tasks and AI Agent workflows. It features high inference throughput and low end-to-end latency, with native support for tool...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/baidu/cobuddy-20260430/endpoints"}},{"id":"openai/gpt-chat-latest","canonical_slug":"openai/gpt-chat-latest-20260505","hugging_face_id":null,"name":"OpenAI: GPT Chat Latest","created":1778000212,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","tool_choice","tools","top_logprobs"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-chat-latest-20260505/endpoints"}},{"id":"x-ai/grok-4.3","canonical_slug":"x-ai/grok-4.3-20260430","hugging_face_id":null,"name":"xAI: Grok 4.3","created":1777591821,"description":"Grok 4.3 is a reasoning model from xAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002"},"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-4.3-20260430/endpoints"}},{"id":"ibm-granite/granite-4.1-8b","canonical_slug":"ibm-granite/granite-4.1-8b-20260429","hugging_face_id":"ibm-granite/granite-4.1-8b","name":"IBM: Granite 4.1 8B","created":1777577071,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000001","input_cache_read":"0.00000005"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/ibm-granite/granite-4.1-8b-20260429/endpoints"}},{"id":"mistralai/mistral-medium-3-5","canonical_slug":"mistralai/mistral-medium-3.5-20260430","hugging_face_id":null,"name":"Mistral: Mistral Medium 3.5","created":1777570439,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-medium-3.5-20260430/endpoints"}},{"id":"openrouter/owl-alpha","canonical_slug":"openrouter/owl-alpha","hugging_face_id":null,"name":"Owl Alpha","created":1777398589,"description":"Owl Alpha is a high-performance foundation model designed for agentic workloads. Natively supports tool use, and long-context tasks, with strong performance in code generation, automated workflows, and complex instruction execution....","context_length":1048756,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048756,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openrouter/owl-alpha/endpoints"}},{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","canonical_slug":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428","hugging_face_id":null,"name":"NVIDIA: Nemotron 3 Nano Omni (free)","created":1777393095,"description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","context_length":256000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":256000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428/endpoints"}},{"id":"poolside/laguna-xs.2:free","canonical_slug":"poolside/laguna-xs.2-20260421","hugging_face_id":"poolside/Laguna-XS.2","name":"Poolside: Laguna XS.2 (free)","created":1777389604,"description":"Laguna XS.2 is the second-generation model in the XS size class from [Poolside](https://poolside.ai), their efficient coding agent series. It combines tool calling and reasoning capabilities with a compact footprint, offering...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"default_parameters":{"temperature":0.7,"top_p":0.9,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/poolside/laguna-xs.2-20260421/endpoints"}},{"id":"poolside/laguna-m.1:free","canonical_slug":"poolside/laguna-m.1-20260312","hugging_face_id":null,"name":"Poolside: Laguna M.1 (free)","created":1777388504,"description":"Laguna M.1 is the flagship coding agent model from [Poolside](https://poolside.ai), optimized for complex software engineering tasks. Designed for agentic coding workflows, it supports tool calling and reasoning, with a 128K...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/poolside/laguna-m.1-20260312/endpoints"}},{"id":"~anthropic/claude-haiku-latest","canonical_slug":"~anthropic/claude-haiku-latest","hugging_face_id":null,"name":"Anthropic Claude Haiku Latest","created":1777318492,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/~anthropic/claude-haiku-latest/endpoints"}},{"id":"~openai/gpt-mini-latest","canonical_slug":"~openai/gpt-mini-latest","hugging_face_id":null,"name":"OpenAI GPT Mini Latest","created":1777318471,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000045","web_search":"0.01","input_cache_read":"0.000000075"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/api/v1/models/~openai/gpt-mini-latest/endpoints"}},{"id":"~google/gemini-pro-latest","canonical_slug":"~google/gemini-pro-latest","hugging_face_id":null,"name":"Google Gemini Pro Latest","created":1777318451,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/~google/gemini-pro-latest/endpoints"}},{"id":"~moonshotai/kimi-latest","canonical_slug":"~moonshotai/kimi-latest","hugging_face_id":null,"name":"MoonshotAI Kimi Latest","created":1777318428,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","context_length":262142,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000073","completion":"0.00000349","input_cache_read":"0.00000025"},"top_provider":{"context_length":262142,"max_completion_tokens":262142,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/~moonshotai/kimi-latest/endpoints"}},{"id":"~google/gemini-flash-latest","canonical_slug":"~google/gemini-flash-latest","hugging_face_id":null,"name":"Google Gemini Flash Latest","created":1777318398,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","image":"0.0000005","audio":"0.000001","web_search":"0.014","internal_reasoning":"0.000003","input_cache_read":"0.00000005","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/~google/gemini-flash-latest/endpoints"}},{"id":"~anthropic/claude-sonnet-latest","canonical_slug":"~anthropic/claude-sonnet-latest","hugging_face_id":null,"name":"Anthropic Claude Sonnet Latest","created":1777318368,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/~anthropic/claude-sonnet-latest/endpoints"}},{"id":"~openai/gpt-latest","canonical_slug":"~openai/gpt-latest","hugging_face_id":null,"name":"OpenAI GPT Latest","created":1777318334,"description":"This model always redirects to the latest model in the OpenAI GPT family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/api/v1/models/~openai/gpt-latest/endpoints"}},{"id":"qwen/qwen3.5-plus-20260420","canonical_slug":"qwen/qwen3.5-plus-20260420","hugging_face_id":null,"name":"Qwen: Qwen3.5 Plus 2026-04-20","created":1777261368,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000018"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-plus-20260420/endpoints"}},{"id":"qwen/qwen3.6-flash","canonical_slug":"qwen/qwen3.6-flash","hugging_face_id":null,"name":"Qwen: Qwen3.6 Flash","created":1777261362,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001875","completion":"0.000001125","input_cache_write":"0.000000234375"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-flash/endpoints"}},{"id":"qwen/qwen3.6-35b-a3b","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","hugging_face_id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-35b-a3b-20260415/endpoints"}},{"id":"qwen/qwen3.6-max-preview","canonical_slug":"qwen/qwen3.6-max-preview-20260420","hugging_face_id":null,"name":"Qwen: Qwen3.6 Max Preview","created":1777260242,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000104","completion":"0.00000624","input_cache_write":"0.0000013"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-max-preview-20260420/endpoints"}},{"id":"qwen/qwen3.6-27b","canonical_slug":"qwen/qwen3.6-27b-20260422","hugging_face_id":"Qwen/Qwen3.6-27B","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.0000032"},"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-27b-20260422/endpoints"}},{"id":"openai/gpt-5.5-pro","canonical_slug":"openai/gpt-5.5-pro-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5 Pro","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00018","web_search":"0.01"},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.5-pro-20260423/endpoints"}},{"id":"openai/gpt-5.5","canonical_slug":"openai/gpt-5.5-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.5-20260423/endpoints"}},{"id":"deepseek/deepseek-v4-pro","canonical_slug":"deepseek/deepseek-v4-pro-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek: DeepSeek V4 Pro","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.000000003625"},"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-pro-20260423/endpoints"}},{"id":"deepseek/deepseek-v4-flash:free","canonical_slug":"deepseek/deepseek-v4-flash-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek: DeepSeek V4 Flash (free)","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260423/endpoints"}},{"id":"deepseek/deepseek-v4-flash","canonical_slug":"deepseek/deepseek-v4-flash-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek: DeepSeek V4 Flash","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000126","completion":"0.000000252","input_cache_read":"0.0000000252"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260423/endpoints"}},{"id":"inclusionai/ling-2.6-1t","canonical_slug":"inclusionai/ling-2.6-1t-20260423","hugging_face_id":null,"name":"inclusionAI: Ling-2.6-1T","created":1776948238,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","input_cache_read":"0.00000006"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/inclusionai/ling-2.6-1t-20260423/endpoints"}},{"id":"tencent/hy3-preview","canonical_slug":"tencent/hy3-preview-20260421","hugging_face_id":"tencent/Hy3-preview","name":"Tencent: Hy3 preview","created":1776878150,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000066","completion":"0.00000026","input_cache_read":"0.000000029"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.9,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/tencent/hy3-preview-20260421/endpoints"}},{"id":"xiaomi/mimo-v2.5-pro","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","hugging_face_id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"Xiaomi: MiMo-V2.5-Pro","created":1776874273,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000003","input_cache_read":"0.0000002"},"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2.5-pro-20260422/endpoints"}},{"id":"xiaomi/mimo-v2.5","canonical_slug":"xiaomi/mimo-v2.5-20260422","hugging_face_id":"XiaomiMiMo/MiMo-V2.5","name":"Xiaomi: MiMo-V2.5","created":1776874269,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1048576,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000008"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2.5-20260422/endpoints"}},{"id":"openai/gpt-5.4-image-2","canonical_slug":"openai/gpt-5.4-image-2-20260421","hugging_face_id":"","name":"OpenAI: GPT-5.4 Image 2","created":1776797528,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000008","completion":"0.000015","web_search":"0.01","input_cache_read":"0.000002"},"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","top_logprobs"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-image-2-20260421/endpoints"}},{"id":"inclusionai/ling-2.6-flash","canonical_slug":"inclusionai/ling-2.6-flash-20260421","hugging_face_id":"","name":"inclusionAI: Ling-2.6-flash","created":1776795886,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0.00000003","input_cache_read":"0.000000002"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/inclusionai/ling-2.6-flash-20260421/endpoints"}},{"id":"~anthropic/claude-opus-latest","canonical_slug":"~anthropic/claude-opus-latest","hugging_face_id":"","name":"Anthropic: Claude Opus Latest","created":1776795361,"description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/~anthropic/claude-opus-latest/endpoints"}},{"id":"openrouter/pareto-code","canonical_slug":"openrouter/pareto-code","hugging_face_id":"","name":"Pareto Code Router","created":1776747900,"description":"The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...","context_length":2000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"-1","completion":"-1"},"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openrouter/pareto-code/endpoints"}},{"id":"baidu/qianfan-ocr-fast","canonical_slug":"baidu/qianfan-ocr-fast-20260420","hugging_face_id":"","name":"Baidu: Qianfan-OCR-Fast","created":1776707472,"description":"Qianfan-OCR-Fast is a domain-specific multimodal large model purpose-built for OCR. By leveraging specialized OCR training data while preserving versatile multimodal intelligence, it provides a powerful performance upgrade over Qianfan-OCR.","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000068","completion":"0.00000281"},"top_provider":{"context_length":65536,"max_completion_tokens":28672,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/baidu/qianfan-ocr-fast-20260420/endpoints"}},{"id":"moonshotai/kimi-k2.6","canonical_slug":"moonshotai/kimi-k2.6-20260420","hugging_face_id":"moonshotai/Kimi-K2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262142,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000073","completion":"0.00000349","input_cache_read":"0.00000025"},"top_provider":{"context_length":262142,"max_completion_tokens":262142,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2.6-20260420/endpoints"}},{"id":"anthropic/claude-opus-4.7","canonical_slug":"anthropic/claude-4.7-opus-20260416","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.7-opus-20260416/endpoints"}},{"id":"anthropic/claude-opus-4.6-fast","canonical_slug":"anthropic/claude-4.6-opus-fast-20260407","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.6 (Fast)","created":1775592472,"description":"Fast-mode variant of [Opus 4.6](/anthropic/claude-opus-4.6) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00015","web_search":"0.01","input_cache_read":"0.000003","input_cache_write":"0.0000375"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-opus-fast-20260407/endpoints"}},{"id":"z-ai/glm-5.1","canonical_slug":"z-ai/glm-5.1-20260406","hugging_face_id":"zai-org/GLM-5.1","name":"Z.ai: GLM 5.1","created":1775578025,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000098","completion":"0.00000308","input_cache_read":"0.000000182"},"top_provider":{"context_length":202752,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.1-20260406/endpoints"}},{"id":"google/gemma-4-26b-a4b-it:free","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","hugging_face_id":"google/gemma-4-26B-A4B-it","name":"Google: Gemma 4 26B A4B (free)","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints"}},{"id":"google/gemma-4-26b-a4b-it","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","hugging_face_id":"google/gemma-4-26B-A4B-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000033"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints"}},{"id":"google/gemma-4-31b-it:free","canonical_slug":"google/gemma-4-31b-it-20260402","hugging_face_id":"google/gemma-4-31B-it","name":"Google: Gemma 4 31B (free)","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-31b-it-20260402/endpoints"}},{"id":"google/gemma-4-31b-it","canonical_slug":"google/gemma-4-31b-it-20260402","hugging_face_id":"google/gemma-4-31B-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.00000012","completion":"0.00000037"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-31b-it-20260402/endpoints"}},{"id":"qwen/qwen3.6-plus","canonical_slug":"qwen/qwen3.6-plus-04-02","hugging_face_id":"","name":"Qwen: Qwen3.6 Plus","created":1775133557,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000325","completion":"0.00000195","input_cache_write":"0.00000040625"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-plus-04-02/endpoints"}},{"id":"z-ai/glm-5v-turbo","canonical_slug":"z-ai/glm-5v-turbo-20260401","hugging_face_id":"","name":"Z.ai: GLM 5V Turbo","created":1775061458,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5v-turbo-20260401/endpoints"}},{"id":"arcee-ai/trinity-large-thinking:free","canonical_slug":"arcee-ai/trinity-large-thinking","hugging_face_id":"arcee-ai/Trinity-Large-Thinking","name":"Arcee AI: Trinity Large Thinking (free)","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":80000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.3,"top_p":0.8,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-large-thinking/endpoints"}},{"id":"arcee-ai/trinity-large-thinking","canonical_slug":"arcee-ai/trinity-large-thinking","hugging_face_id":"arcee-ai/Trinity-Large-Thinking","name":"Arcee AI: Trinity Large Thinking","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000022","completion":"0.00000085","input_cache_read":"0.00000006"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.3,"top_p":0.8,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-large-thinking/endpoints"}},{"id":"x-ai/grok-4.20-multi-agent","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","hugging_face_id":"","name":"xAI: Grok 4.20 Multi-Agent","created":1774979158,"description":"Grok 4.20 Multi-Agent is a variant of xAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000002"},"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-01","expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-4.20-multi-agent-20260309/endpoints"}},{"id":"x-ai/grok-4.20","canonical_slug":"x-ai/grok-4.20-20260309","hugging_face_id":"","name":"xAI: Grok 4.20","created":1774979019,"description":"Grok 4.20 is a reasoning model from xAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002"},"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-01","expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-4.20-20260309/endpoints"}},{"id":"google/lyria-3-pro-preview","canonical_slug":"google/lyria-3-pro-preview-20260330","hugging_face_id":null,"name":"Google: Lyria 3 Pro Preview","created":1774907286,"description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/lyria-3-pro-preview-20260330/endpoints"}},{"id":"google/lyria-3-clip-preview","canonical_slug":"google/lyria-3-clip-preview-20260330","hugging_face_id":null,"name":"Google: Lyria 3 Clip Preview","created":1774907255,"description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/lyria-3-clip-preview-20260330/endpoints"}},{"id":"kwaipilot/kat-coder-pro-v2","canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","hugging_face_id":"","name":"Kwaipilot: KAT-Coder-Pro V2","created":1774649310,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006"},"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/kwaipilot/kat-coder-pro-v2-20260327/endpoints"}},{"id":"rekaai/reka-edge","canonical_slug":"rekaai/reka-edge-2603","hugging_face_id":"RekaAI/reka-edge-2603","name":"Reka Edge","created":1774026965,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000001"},"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/rekaai/reka-edge-2603/endpoints"}},{"id":"xiaomi/mimo-v2-omni","canonical_slug":"xiaomi/mimo-v2-omni-20260318","hugging_face_id":"","name":"Xiaomi: MiMo-V2-Omni","created":1773863703,"description":"MiMo-V2-Omni is a frontier omni-modal model that natively processes image, video, and audio inputs within a unified architecture. It combines strong multimodal perception with agentic capability - visual grounding, multi-step...","context_length":262144,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000008"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2-omni-20260318/endpoints"}},{"id":"xiaomi/mimo-v2-pro","canonical_slug":"xiaomi/mimo-v2-pro-20260318","hugging_face_id":"","name":"Xiaomi: MiMo-V2-Pro","created":1773863643,"description":"MiMo-V2-Pro is Xiaomi's flagship foundation model, featuring over 1T total parameters and a 1M context length, deeply optimized for agentic scenarios. It is highly adaptable to general agent frameworks like...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000003","input_cache_read":"0.0000002"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2-pro-20260318/endpoints"}},{"id":"minimax/minimax-m2.7","canonical_slug":"minimax/minimax-m2.7-20260318","hugging_face_id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax: MiniMax M2.7","created":1773836697,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":196608,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000279","completion":"0.0000012"},"top_provider":{"context_length":196608,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.7-20260318/endpoints"}},{"id":"openai/gpt-5.4-nano","canonical_slug":"openai/gpt-5.4-nano-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.00000125","web_search":"0.01","input_cache_read":"0.00000002"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-nano-20260317/endpoints"}},{"id":"openai/gpt-5.4-mini","canonical_slug":"openai/gpt-5.4-mini-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000045","web_search":"0.01","input_cache_read":"0.000000075"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-mini-20260317/endpoints"}},{"id":"mistralai/mistral-small-2603","canonical_slug":"mistralai/mistral-small-2603","hugging_face_id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-2603/endpoints"}},{"id":"z-ai/glm-5-turbo","canonical_slug":"z-ai/glm-5-turbo-20260315","hugging_face_id":"","name":"Z.ai: GLM 5 Turbo","created":1773583573,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5-turbo-20260315/endpoints"}},{"id":"nvidia/nemotron-3-super-120b-a12b:free","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"NVIDIA: Nemotron 3 Super (free)","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints"}},{"id":"nvidia/nemotron-3-super-120b-a12b","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000045"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints"}},{"id":"bytedance-seed/seed-2.0-lite","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","hugging_face_id":null,"name":"ByteDance Seed: Seed-2.0-Lite","created":1773157231,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002"},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-2.0-lite-20260309/endpoints"}},{"id":"qwen/qwen3.5-9b","canonical_slug":"qwen/qwen3.5-9b-20260310","hugging_face_id":"Qwen/Qwen3.5-9B","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000004","completion":"0.00000015"},"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-9b-20260310/endpoints"}},{"id":"openai/gpt-5.4-pro","canonical_slug":"openai/gpt-5.4-pro-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4 Pro","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00018","web_search":"0.01"},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-pro-20260305/endpoints"}},{"id":"openai/gpt-5.4","canonical_slug":"openai/gpt-5.4-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-20260305/endpoints"}},{"id":"inception/mercury-2","canonical_slug":"inception/mercury-2-20260304","hugging_face_id":null,"name":"Inception: Mercury 2","created":1772636275,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.00000075","input_cache_read":"0.000000025"},"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"default_parameters":{"temperature":0.75,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/inception/mercury-2-20260304/endpoints"}},{"id":"openai/gpt-5.3-chat","canonical_slug":"openai/gpt-5.3-chat-20260303","hugging_face_id":"","name":"OpenAI: GPT-5.3 Chat","created":1772564061,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.3-chat-20260303/endpoints"}},{"id":"google/gemini-3.1-flash-lite-preview","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","hugging_face_id":"","name":"Google: Gemini 3.1 Flash Lite Preview","created":1772512673,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.1-flash-lite-preview-20260303/endpoints"}},{"id":"bytedance-seed/seed-2.0-mini","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","hugging_face_id":"","name":"ByteDance Seed: Seed-2.0-Mini","created":1772131107,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004"},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-2.0-mini-20260224/endpoints"}},{"id":"google/gemini-3.1-flash-image-preview","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","hugging_face_id":"","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","created":1772119558,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","web_search":"0.014"},"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.1-flash-image-preview-20260226/endpoints"}},{"id":"qwen/qwen3.5-35b-a3b","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","hugging_face_id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.000001","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-35b-a3b-20260224/endpoints"}},{"id":"qwen/qwen3.5-27b","canonical_slug":"qwen/qwen3.5-27b-20260224","hugging_face_id":"Qwen/Qwen3.5-27B","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000195","completion":"0.00000156"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-27b-20260224/endpoints"}},{"id":"qwen/qwen3.5-122b-a10b","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","hugging_face_id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen: Qwen3.5-122B-A10B","created":1772053789,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000208"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-122b-a10b-20260224/endpoints"}},{"id":"qwen/qwen3.5-flash-02-23","canonical_slug":"qwen/qwen3.5-flash-20260224","hugging_face_id":null,"name":"Qwen: Qwen3.5-Flash","created":1772053776,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000065","completion":"0.00000026","input_cache_write":"0.00000008125"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-flash-20260224/endpoints"}},{"id":"liquid/lfm-2-24b-a2b","canonical_slug":"liquid/lfm-2-24b-a2b-20260224","hugging_face_id":"LiquidAI/LFM2-24B-A2B","name":"LiquidAI: LFM2-24B-A2B","created":1772048711,"description":"LFM2-24B-A2B is the largest model in the LFM2 family of hybrid architectures designed for efficient on-device deployment. Built as a 24B parameter Mixture-of-Experts model with only 2B active parameters per...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000012"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":0.1,"top_p":null,"top_k":50,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1.05},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/liquid/lfm-2-24b-a2b-20260224/endpoints"}},{"id":"google/gemini-3.1-pro-preview-customtools","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","hugging_face_id":null,"name":"Google: Gemini 3.1 Pro Preview Custom Tools","created":1772045923,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.1-pro-preview-customtools-20260219/endpoints"}},{"id":"openai/gpt-5.3-codex","canonical_slug":"openai/gpt-5.3-codex-20260224","hugging_face_id":"","name":"OpenAI: GPT-5.3-Codex","created":1771959164,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.3-codex-20260224/endpoints"}},{"id":"aion-labs/aion-2.0","canonical_slug":"aion-labs/aion-2.0-20260223","hugging_face_id":null,"name":"AionLabs: Aion-2.0","created":1771881306,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000016","input_cache_read":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/aion-labs/aion-2.0-20260223/endpoints"}},{"id":"google/gemini-3.1-pro-preview","canonical_slug":"google/gemini-3.1-pro-preview-20260219","hugging_face_id":"","name":"Google: Gemini 3.1 Pro Preview","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.1-pro-preview-20260219/endpoints"}},{"id":"anthropic/claude-sonnet-4.6","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-sonnet-20260217/endpoints"}},{"id":"qwen/qwen3.5-plus-02-15","canonical_slug":"qwen/qwen3.5-plus-20260216","hugging_face_id":"","name":"Qwen: Qwen3.5 Plus 2026-02-15","created":1771229416,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000156","input_cache_write":"0.000000325"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-plus-20260216/endpoints"}},{"id":"qwen/qwen3.5-397b-a17b","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","hugging_face_id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000039","completion":"0.00000234","input_cache_read":"0.000000195"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-397b-a17b-20260216/endpoints"}},{"id":"minimax/minimax-m2.5:free","canonical_slug":"minimax/minimax-m2.5-20260211","hugging_face_id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax: MiniMax M2.5 (free)","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":196608,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":196608,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","temperature","tools"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.5-20260211/endpoints"}},{"id":"minimax/minimax-m2.5","canonical_slug":"minimax/minimax-m2.5-20260211","hugging_face_id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax: MiniMax M2.5","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":196608,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000115"},"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.5-20260211/endpoints"}},{"id":"z-ai/glm-5","canonical_slug":"z-ai/glm-5-20260211","hugging_face_id":"zai-org/GLM-5","name":"Z.ai: GLM 5","created":1770829182,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012"},"top_provider":{"context_length":202752,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5-20260211/endpoints"}},{"id":"qwen/qwen3-max-thinking","canonical_slug":"qwen/qwen3-max-thinking-20260123","hugging_face_id":null,"name":"Qwen: Qwen3 Max Thinking","created":1770671901,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000078","completion":"0.0000039"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-max-thinking-20260123/endpoints"}},{"id":"anthropic/claude-opus-4.6","canonical_slug":"anthropic/claude-4.6-opus-20260205","hugging_face_id":"","name":"Anthropic: Claude Opus 4.6","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-opus-20260205/endpoints"}},{"id":"qwen/qwen3-coder-next","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","hugging_face_id":"Qwen/Qwen3-Coder-Next","name":"Qwen: Qwen3 Coder Next","created":1770164101,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000011","completion":"0.0000008","input_cache_read":"0.00000007"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-next-2025-02-03/endpoints"}},{"id":"openrouter/free","canonical_slug":"openrouter/free","hugging_face_id":"","name":"Free Models Router","created":1769917427,"description":"The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openrouter/free/endpoints"}},{"id":"stepfun/step-3.5-flash","canonical_slug":"stepfun/step-3.5-flash","hugging_face_id":"stepfun-ai/Step-3.5-Flash","name":"StepFun: Step 3.5 Flash","created":1769728337,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/stepfun/step-3.5-flash/endpoints"}},{"id":"arcee-ai/trinity-large-preview","canonical_slug":"arcee-ai/trinity-large-preview","hugging_face_id":"arcee-ai/Trinity-Large-Preview","name":"Arcee AI: Trinity Large Preview","created":1769552670,"description":"Trinity-Large-Preview is a frontier-scale open-weight language model from Arcee, built as a 400B-parameter sparse Mixture-of-Experts with 13B active parameters per token using 4-of-256 expert routing. It excels in creative writing,...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000045"},"top_provider":{"context_length":131000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","structured_outputs","temperature","tools","top_k","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.8,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-large-preview/endpoints"}},{"id":"moonshotai/kimi-k2.5","canonical_slug":"moonshotai/kimi-k2.5-0127","hugging_face_id":"moonshotai/Kimi-K2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.0000019","input_cache_read":"0.00000009"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2.5-0127/endpoints"}},{"id":"upstage/solar-pro-3","canonical_slug":"upstage/solar-pro-3","hugging_face_id":"","name":"Upstage: Solar Pro 3","created":1769481200,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/upstage/solar-pro-3/endpoints"}},{"id":"minimax/minimax-m2-her","canonical_slug":"minimax/minimax-m2-her-20260123","hugging_face_id":"","name":"MiniMax: MiniMax M2-her","created":1769177239,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2-her-20260123/endpoints"}},{"id":"writer/palmyra-x5","canonical_slug":"writer/palmyra-x5-20250428","hugging_face_id":"","name":"Writer: Palmyra X5","created":1769003823,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.000006"},"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/writer/palmyra-x5-20250428/endpoints"}},{"id":"liquid/lfm-2.5-1.2b-thinking:free","canonical_slug":"liquid/lfm-2.5-1.2b-thinking-20260120","hugging_face_id":"LiquidAI/LFM2.5-1.2B-Thinking","name":"LiquidAI: LFM2.5-1.2B-Thinking (free)","created":1768927527,"description":"LFM2.5-1.2B-Thinking is a lightweight reasoning-focused model optimized for agentic tasks, data extraction, and RAG—while still running comfortably on edge devices. It supports long context (up to 32K tokens) and is...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/liquid/lfm-2.5-1.2b-thinking-20260120/endpoints"}},{"id":"liquid/lfm-2.5-1.2b-instruct:free","canonical_slug":"liquid/lfm-2.5-1.2b-instruct-20260120","hugging_face_id":"LiquidAI/LFM2.5-1.2B-Instruct","name":"LiquidAI: LFM2.5-1.2B-Instruct (free)","created":1768927521,"description":"LFM2.5-1.2B-Instruct is a compact, high-performance instruction-tuned model built for fast on-device AI. It delivers strong chat quality in a 1.2B parameter footprint, with efficient edge inference and broad runtime support.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/liquid/lfm-2.5-1.2b-instruct-20260120/endpoints"}},{"id":"openai/gpt-audio","canonical_slug":"openai/gpt-audio","hugging_face_id":"","name":"OpenAI: GPT Audio","created":1768862569,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","audio":"0.000032"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-audio/endpoints"}},{"id":"openai/gpt-audio-mini","canonical_slug":"openai/gpt-audio-mini","hugging_face_id":"","name":"OpenAI: GPT Audio Mini","created":1768859419,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000024","audio":"0.0000006"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-audio-mini/endpoints"}},{"id":"z-ai/glm-4.7-flash","canonical_slug":"z-ai/glm-4.7-flash-20260119","hugging_face_id":"zai-org/GLM-4.7-Flash","name":"Z.ai: GLM 4.7 Flash","created":1768833913,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.0000004","input_cache_read":"0.00000001"},"top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.7-flash-20260119/endpoints"}},{"id":"openai/gpt-5.2-codex","canonical_slug":"openai/gpt-5.2-codex-20260114","hugging_face_id":"","name":"OpenAI: GPT-5.2-Codex","created":1768409315,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","input_cache_read":"0.000000175"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-codex-20260114/endpoints"}},{"id":"bytedance-seed/seed-1.6-flash","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","hugging_face_id":"","name":"ByteDance Seed: Seed 1.6 Flash","created":1766505011,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-1.6-flash-20250625/endpoints"}},{"id":"bytedance-seed/seed-1.6","canonical_slug":"bytedance-seed/seed-1.6-20250625","hugging_face_id":"","name":"ByteDance Seed: Seed 1.6","created":1766504997,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-1.6-20250625/endpoints"}},{"id":"minimax/minimax-m2.1","canonical_slug":"minimax/minimax-m2.1","hugging_face_id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax: MiniMax M2.1","created":1766454997,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":196608,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000029","completion":"0.00000095","input_cache_read":"0.00000003"},"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.9,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.1/endpoints"}},{"id":"z-ai/glm-4.7","canonical_slug":"z-ai/glm-4.7-20251222","hugging_face_id":"zai-org/GLM-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.00000175","input_cache_read":"0.00000008"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.7-20251222/endpoints"}},{"id":"google/gemini-3-flash-preview","canonical_slug":"google/gemini-3-flash-preview-20251217","hugging_face_id":"","name":"Google: Gemini 3 Flash Preview","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","image":"0.0000005","audio":"0.000001","web_search":"0.014","internal_reasoning":"0.000003","input_cache_read":"0.00000005","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3-flash-preview-20251217/endpoints"}},{"id":"xiaomi/mimo-v2-flash","canonical_slug":"xiaomi/mimo-v2-flash-20251210","hugging_face_id":"XiaomiMiMo/MiMo-V2-Flash","name":"Xiaomi: MiMo-V2-Flash","created":1765731308,"description":"MiMo-V2-Flash is an open-source foundation language model developed by Xiaomi. It is a Mixture-of-Experts model with 309B total parameters and 15B active parameters, adopting hybrid attention architecture. MiMo-V2-Flash supports a...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003","input_cache_read":"0.00000001"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2-flash-20251210/endpoints"}},{"id":"nvidia/nemotron-3-nano-30b-a3b:free","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","name":"NVIDIA: Nemotron 3 Nano 30B A3B (free)","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints"}},{"id":"nvidia/nemotron-3-nano-30b-a3b","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","name":"NVIDIA: Nemotron 3 Nano 30B A3B","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000002"},"top_provider":{"context_length":262144,"max_completion_tokens":228000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints"}},{"id":"openai/gpt-5.2-chat","canonical_slug":"openai/gpt-5.2-chat-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2 Chat","created":1765389783,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","input_cache_read":"0.000000175"},"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-chat-20251211/endpoints"}},{"id":"openai/gpt-5.2-pro","canonical_slug":"openai/gpt-5.2-pro-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2 Pro","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000021","completion":"0.000168","web_search":"0.01"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-pro-20251211/endpoints"}},{"id":"openai/gpt-5.2","canonical_slug":"openai/gpt-5.2-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","input_cache_read":"0.000000175"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-20251211/endpoints"}},{"id":"mistralai/devstral-2512","canonical_slug":"mistralai/devstral-2512","hugging_face_id":"mistralai/Devstral-2-123B-Instruct-2512","name":"Mistral: Devstral 2 2512","created":1765285419,"description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","context_length":262144,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/devstral-2512/endpoints"}},{"id":"relace/relace-search","canonical_slug":"relace/relace-search-20251208","hugging_face_id":null,"name":"Relace: Relace Search","created":1765213560,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000003"},"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","seed","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/relace/relace-search-20251208/endpoints"}},{"id":"z-ai/glm-4.6v","canonical_slug":"z-ai/glm-4.6-20251208","hugging_face_id":"zai-org/GLM-4.6V","name":"Z.ai: GLM 4.6V","created":1765207462,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000005"},"top_provider":{"context_length":131072,"max_completion_tokens":24000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.6,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.6-20251208/endpoints"}},{"id":"nex-agi/deepseek-v3.1-nex-n1","canonical_slug":"nex-agi/deepseek-v3.1-nex-n1","hugging_face_id":"nex-agi/DeepSeek-V3.1-Nex-N1","name":"Nex AGI: DeepSeek V3.1 Nex N1","created":1765204393,"description":"DeepSeek V3.1 Nex-N1 is the flagship release of the Nex-N1 series — a post-trained model designed to highlight agent autonomy, tool use, and real-world productivity. Nex-N1 demonstrates competitive performance across...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000135","completion":"0.0000005"},"top_provider":{"context_length":131072,"max_completion_tokens":163840,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nex-agi/deepseek-v3.1-nex-n1/endpoints"}},{"id":"essentialai/rnj-1-instruct","canonical_slug":"essentialai/rnj-1-instruct","hugging_face_id":"EssentialAI/rnj-1-instruct","name":"EssentialAI: Rnj 1 Instruct","created":1765094847,"description":"Rnj-1 is an 8B-parameter, dense, open-weight model family developed by Essential AI and trained from scratch with a focus on programming, math, and scientific reasoning. The model demonstrates strong performance...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000015"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/essentialai/rnj-1-instruct/endpoints"}},{"id":"openrouter/bodybuilder","canonical_slug":"openrouter/bodybuilder","hugging_face_id":"","name":"Body Builder (beta)","created":1764903653,"description":"Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"-1","completion":"-1"},"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openrouter/bodybuilder/endpoints"}},{"id":"openai/gpt-5.1-codex-max","canonical_slug":"openai/gpt-5.1-codex-max-20251204","hugging_face_id":"","name":"OpenAI: GPT-5.1-Codex-Max","created":1764878934,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-codex-max-20251204/endpoints"}},{"id":"amazon/nova-2-lite-v1","canonical_slug":"amazon/nova-2-lite-v1","hugging_face_id":"","name":"Amazon: Nova 2 Lite","created":1764696672,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025"},"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-2-lite-v1/endpoints"}},{"id":"mistralai/ministral-14b-2512","canonical_slug":"mistralai/ministral-14b-2512","hugging_face_id":"mistralai/Ministral-3-14B-Instruct-2512","name":"Mistral: Ministral 3 14B 2512","created":1764681735,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/ministral-14b-2512/endpoints"}},{"id":"mistralai/ministral-8b-2512","canonical_slug":"mistralai/ministral-8b-2512","hugging_face_id":"mistralai/Ministral-3-8B-Instruct-2512","name":"Mistral: Ministral 3 8B 2512","created":1764681654,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/ministral-8b-2512/endpoints"}},{"id":"mistralai/ministral-3b-2512","canonical_slug":"mistralai/ministral-3b-2512","hugging_face_id":"mistralai/Ministral-3-3B-Instruct-2512","name":"Mistral: Ministral 3 3B 2512","created":1764681560,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/ministral-3b-2512/endpoints"}},{"id":"mistralai/mistral-large-2512","canonical_slug":"mistralai/mistral-large-2512","hugging_face_id":"","name":"Mistral: Mistral Large 3 2512","created":1764624472,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000015","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.0645,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-large-2512/endpoints"}},{"id":"arcee-ai/trinity-mini","canonical_slug":"arcee-ai/trinity-mini-20251201","hugging_face_id":"arcee-ai/Trinity-Mini","name":"Arcee AI: Trinity Mini","created":1764601720,"description":"Trinity Mini is a 26B-parameter (3B active) sparse mixture-of-experts language model featuring 128 experts with 8 active per token. Engineered for efficient reasoning over long contexts (131k) with robust function...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000045","completion":"0.00000015"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.15,"top_p":0.75,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-mini-20251201/endpoints"}},{"id":"deepseek/deepseek-v3.2-speciale","canonical_slug":"deepseek/deepseek-v3.2-speciale-20251201","hugging_face_id":"deepseek-ai/DeepSeek-V3.2-Speciale","name":"DeepSeek: DeepSeek V3.2 Speciale","created":1764594837,"description":"DeepSeek-V3.2-Speciale is a high-compute variant of DeepSeek-V3.2 optimized for maximum reasoning and agentic performance. It builds on DeepSeek Sparse Attention (DSA) for efficient long-context processing, then scales post-training reinforcement learning...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000287","completion":"0.000000431","input_cache_read":"0.000000058"},"top_provider":{"context_length":163840,"max_completion_tokens":163840,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-speciale-20251201/endpoints"}},{"id":"deepseek/deepseek-v3.2","canonical_slug":"deepseek/deepseek-v3.2-20251201","hugging_face_id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000252","completion":"0.000000378","input_cache_read":"0.0000000252"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-20251201/endpoints"}},{"id":"prime-intellect/intellect-3","canonical_slug":"prime-intellect/intellect-3-20251126","hugging_face_id":"PrimeIntellect/INTELLECT-3-FP8","name":"Prime Intellect: INTELLECT-3","created":1764212534,"description":"INTELLECT-3 is a 106B-parameter Mixture-of-Experts model (12B active) post-trained from GLM-4.5-Air-Base using supervised fine-tuning (SFT) followed by large-scale reinforcement learning (RL). It offers state-of-the-art performance for its size across math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000011"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/prime-intellect/intellect-3-20251126/endpoints"}},{"id":"anthropic/claude-opus-4.5","canonical_slug":"anthropic/claude-4.5-opus-20251124","hugging_face_id":"","name":"Anthropic: Claude Opus 4.5","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-opus-20251124/endpoints"}},{"id":"allenai/olmo-3-32b-think","canonical_slug":"allenai/olmo-3-32b-think-20251121","hugging_face_id":"allenai/Olmo-3-32B-Think","name":"AllenAI: Olmo 3 32B Think","created":1763758276,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000005"},"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/allenai/olmo-3-32b-think-20251121/endpoints"}},{"id":"google/gemini-3-pro-image-preview","canonical_slug":"google/gemini-3-pro-image-preview-20251120","hugging_face_id":"","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","created":1763653797,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3-pro-image-preview-20251120/endpoints"}},{"id":"x-ai/grok-4.1-fast","canonical_slug":"x-ai/grok-4.1-fast","hugging_face_id":"","name":"xAI: Grok 4.1 Fast","created":1763587502,"description":"Grok 4.1 Fast is xAI's best agentic tool calling model that shines in real-world use cases like customer support and deep research. 2M context window. Reasoning can be enabled/disabled using...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000005","web_search":"0.005","input_cache_read":"0.00000005"},"top_provider":{"context_length":2000000,"max_completion_tokens":30000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-4.1-fast/endpoints"}},{"id":"deepcogito/cogito-v2.1-671b","canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","hugging_face_id":"","name":"Deep Cogito: Cogito v2.1 671B","created":1763071233,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00000125"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepcogito/cogito-v2.1-671b-20251118/endpoints"}},{"id":"openai/gpt-5.1","canonical_slug":"openai/gpt-5.1-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","input_cache_read":"0.00000013"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-20251113/endpoints"}},{"id":"openai/gpt-5.1-chat","canonical_slug":"openai/gpt-5.1-chat-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1 Chat","created":1763060302,"description":"GPT-5.1 Chat (AKA Instant is the fast, lightweight member of the 5.1 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-chat-20251113/endpoints"}},{"id":"openai/gpt-5.1-codex","canonical_slug":"openai/gpt-5.1-codex-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1-Codex","created":1763060298,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-codex-20251113/endpoints"}},{"id":"openai/gpt-5.1-codex-mini","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1-Codex-Mini","created":1763057820,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002","input_cache_read":"0.00000003"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-codex-mini-20251113/endpoints"}},{"id":"moonshotai/kimi-k2-thinking","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","hugging_face_id":"moonshotai/Kimi-K2-Thinking","name":"MoonshotAI: Kimi K2 Thinking","created":1762440622,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000025","input_cache_read":"0.00000015"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2-thinking-20251106/endpoints"}},{"id":"amazon/nova-premier-v1","canonical_slug":"amazon/nova-premier-v1","hugging_face_id":"","name":"Amazon: Nova Premier 1.0","created":1761950332,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","input_cache_read":"0.000000625"},"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-premier-v1/endpoints"}},{"id":"perplexity/sonar-pro-search","canonical_slug":"perplexity/sonar-pro-search","hugging_face_id":"","name":"Perplexity: Sonar Pro Search","created":1761854366,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.018"},"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p","web_search_options"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/perplexity/sonar-pro-search/endpoints"}},{"id":"mistralai/voxtral-small-24b-2507","canonical_slug":"mistralai/voxtral-small-24b-2507","hugging_face_id":"mistralai/Voxtral-Small-24B-2507","name":"Mistral: Voxtral Small 24B 2507","created":1761835144,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001"},"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.2,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/voxtral-small-24b-2507/endpoints"}},{"id":"openai/gpt-oss-safeguard-20b","canonical_slug":"openai/gpt-oss-safeguard-20b","hugging_face_id":"openai/gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","created":1761752836,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003","input_cache_read":"0.000000037"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-safeguard-20b/endpoints"}},{"id":"nvidia/nemotron-nano-12b-v2-vl:free","canonical_slug":"nvidia/nemotron-nano-12b-v2-vl","hugging_face_id":"nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16","name":"NVIDIA: Nemotron Nano 12B 2 VL (free)","created":1761675565,"description":"NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...","context_length":128000,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-nano-12b-v2-vl/endpoints"}},{"id":"minimax/minimax-m2","canonical_slug":"minimax/minimax-m2","hugging_face_id":"MiniMaxAI/MiniMax-M2","name":"MiniMax: MiniMax M2","created":1761252093,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":196608,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000255","completion":"0.000001","input_cache_read":"0.00000003"},"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2/endpoints"}},{"id":"qwen/qwen3-vl-32b-instruct","canonical_slug":"qwen/qwen3-vl-32b-instruct","hugging_face_id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000000104","completion":"0.000000416"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-32b-instruct/endpoints"}},{"id":"ibm-granite/granite-4.0-h-micro","canonical_slug":"ibm-granite/granite-4.0-h-micro","hugging_face_id":"ibm-granite/granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","created":1760927695,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000017","completion":"0.00000011"},"top_provider":{"context_length":131000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/ibm-granite/granite-4.0-h-micro/endpoints"}},{"id":"microsoft/phi-4-mini-instruct","canonical_slug":"microsoft/phi-4-mini-instruct","hugging_face_id":"microsoft/Phi-4-mini-instruct","name":"Microsoft: Phi 4 Mini Instruct","created":1760726049,"description":"Phi-4-mini-instruct is a lightweight open model built upon synthetic data and filtered publicly available websites - with a focus on high-quality, reasoning dense data. The model belongs to the Phi-4...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.00000035","input_cache_read":"0.00000008"},"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/microsoft/phi-4-mini-instruct/endpoints"}},{"id":"openai/gpt-5-image-mini","canonical_slug":"openai/gpt-5-image-mini","hugging_face_id":"","name":"OpenAI: GPT-5 Image Mini","created":1760624583,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-image-mini/endpoints"}},{"id":"anthropic/claude-haiku-4.5","canonical_slug":"anthropic/claude-4.5-haiku-20251001","hugging_face_id":"","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-haiku-20251001/endpoints"}},{"id":"qwen/qwen3-vl-8b-thinking","canonical_slug":"qwen/qwen3-vl-8b-thinking","hugging_face_id":"Qwen/Qwen3-VL-8B-Thinking","name":"Qwen: Qwen3 VL 8B Thinking","created":1760463746,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000117","completion":"0.000001365"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-8b-thinking/endpoints"}},{"id":"qwen/qwen3-vl-8b-instruct","canonical_slug":"qwen/qwen3-vl-8b-instruct","hugging_face_id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen: Qwen3 VL 8B Instruct","created":1760463308,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.0000005"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-8b-instruct/endpoints"}},{"id":"openai/gpt-5-image","canonical_slug":"openai/gpt-5-image","hugging_face_id":"","name":"OpenAI: GPT-5 Image","created":1760447986,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00001","web_search":"0.01","input_cache_read":"0.00000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-image/endpoints"}},{"id":"openai/o3-deep-research","canonical_slug":"openai/o3-deep-research-2025-06-26","hugging_face_id":"","name":"OpenAI: o3 Deep Research","created":1760129661,"description":"o3-deep-research is OpenAI's advanced model for deep research, designed to tackle complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00004","web_search":"0.01","input_cache_read":"0.0000025"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/o3-deep-research-2025-06-26/endpoints"}},{"id":"openai/o4-mini-deep-research","canonical_slug":"openai/o4-mini-deep-research-2025-06-26","hugging_face_id":"","name":"OpenAI: o4 Mini Deep Research","created":1760129642,"description":"o4-mini-deep-research is OpenAI's faster, more affordable deep research model—ideal for tackling complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/o4-mini-deep-research-2025-06-26/endpoints"}},{"id":"nvidia/llama-3.3-nemotron-super-49b-v1.5","canonical_slug":"nvidia/llama-3.3-nemotron-super-49b-v1.5","hugging_face_id":"nvidia/Llama-3_3-Nemotron-Super-49B-v1_5","name":"NVIDIA: Llama 3.3 Nemotron Super 49B V1.5","created":1760101395,"description":"Llama-3.3-Nemotron-Super-49B-v1.5 is a 49B-parameter, English-centric reasoning/chat model derived from Meta’s Llama-3.3-70B-Instruct with a 128K context. It’s post-trained for agentic workflows (RAG, tool calling) via SFT across math, code, science, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/api/v1/models/nvidia/llama-3.3-nemotron-super-49b-v1.5/endpoints"}},{"id":"baidu/ernie-4.5-21b-a3b-thinking","canonical_slug":"baidu/ernie-4.5-21b-a3b-thinking","hugging_face_id":"baidu/ERNIE-4.5-21B-A3B-Thinking","name":"Baidu: ERNIE 4.5 21B A3B Thinking","created":1760048887,"description":"ERNIE-4.5-21B-A3B-Thinking is Baidu's upgraded lightweight MoE model, refined to boost reasoning depth and quality for top-tier performance in logical puzzles, math, science, coding, text generation, and expert-level academic benchmarks.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.00000028"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/baidu/ernie-4.5-21b-a3b-thinking/endpoints"}},{"id":"google/gemini-2.5-flash-image","canonical_slug":"google/gemini-2.5-flash-image","hugging_face_id":"","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","created":1759870431,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","audio":"0.000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-flash-image/endpoints"}},{"id":"qwen/qwen3-vl-30b-a3b-thinking","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","created":1759794479,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000013","completion":"0.00000156"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-30b-a3b-thinking/endpoints"}},{"id":"qwen/qwen3-vl-30b-a3b-instruct","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000013","completion":"0.00000052"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-30b-a3b-instruct/endpoints"}},{"id":"openai/gpt-5-pro","canonical_slug":"openai/gpt-5-pro-2025-10-06","hugging_face_id":"","name":"OpenAI: GPT-5 Pro","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.00012","web_search":"0.01"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-pro-2025-10-06/endpoints"}},{"id":"z-ai/glm-4.6","canonical_slug":"z-ai/glm-4.6","hugging_face_id":"zai-org/GLM-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000043","completion":"0.00000174","input_cache_read":"0.00000008"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.6/endpoints"}},{"id":"anthropic/claude-sonnet-4.5","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375"},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-sonnet-20250929/endpoints"}},{"id":"deepseek/deepseek-v3.2-exp","canonical_slug":"deepseek/deepseek-v3.2-exp","hugging_face_id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"DeepSeek: DeepSeek V3.2 Exp","created":1759150481,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.00000027","completion":"0.00000041"},"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-exp/endpoints"}},{"id":"thedrummer/cydonia-24b-v4.1","canonical_slug":"thedrummer/cydonia-24b-v4.1","hugging_face_id":"thedrummer/cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","created":1758931878,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000005","input_cache_read":"0.00000015"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/thedrummer/cydonia-24b-v4.1/endpoints"}},{"id":"relace/relace-apply-3","canonical_slug":"relace/relace-apply-3","hugging_face_id":"","name":"Relace: Relace Apply 3","created":1758891572,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000085","completion":"0.00000125"},"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","seed","stop"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/relace/relace-apply-3/endpoints"}},{"id":"google/gemini-2.5-flash-lite-preview-09-2025","canonical_slug":"google/gemini-2.5-flash-lite-preview-09-2025","hugging_face_id":"","name":"Google: Gemini 2.5 Flash Lite Preview 09-2025","created":1758819686,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","image":"0.0000001","audio":"0.0000003","web_search":"0.014","internal_reasoning":"0.0000004","input_cache_read":"0.00000001","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-flash-lite-preview-09-2025/endpoints"}},{"id":"qwen/qwen3-vl-235b-a22b-thinking","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","created":1758668690,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.0000026"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-235b-a22b-thinking/endpoints"}},{"id":"qwen/qwen3-vl-235b-a22b-instruct","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.00000088","input_cache_read":"0.00000011"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-235b-a22b-instruct/endpoints"}},{"id":"qwen/qwen3-max","canonical_slug":"qwen/qwen3-max","hugging_face_id":"","name":"Qwen: Qwen3 Max","created":1758662808,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000078","completion":"0.0000039","input_cache_read":"0.000000156","input_cache_write":"0.000000975"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":1,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-max/endpoints"}},{"id":"qwen/qwen3-coder-plus","canonical_slug":"qwen/qwen3-coder-plus","hugging_face_id":"","name":"Qwen: Qwen3 Coder Plus","created":1758662707,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000065","completion":"0.00000325","input_cache_read":"0.00000013","input_cache_write":"0.0000008125"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-plus/endpoints"}},{"id":"openai/gpt-5-codex","canonical_slug":"openai/gpt-5-codex","hugging_face_id":"","name":"OpenAI: GPT-5 Codex","created":1758643403,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-codex/endpoints"}},{"id":"deepseek/deepseek-v3.1-terminus","canonical_slug":"deepseek/deepseek-v3.1-terminus","hugging_face_id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","created":1758548275,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.00000027","completion":"0.00000095","input_cache_read":"0.00000013"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.1-terminus/endpoints"}},{"id":"x-ai/grok-4-fast","canonical_slug":"x-ai/grok-4-fast","hugging_face_id":"","name":"xAI: Grok 4 Fast","created":1758240090,"description":"Grok 4 Fast is xAI's latest multimodal model with SOTA cost-efficiency and a 2M token context window. It comes in two flavors: non-reasoning and reasoning. Read more about the model...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000005","web_search":"0.005","input_cache_read":"0.00000005"},"top_provider":{"context_length":2000000,"max_completion_tokens":30000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-4-fast/endpoints"}},{"id":"alibaba/tongyi-deepresearch-30b-a3b","canonical_slug":"alibaba/tongyi-deepresearch-30b-a3b","hugging_face_id":"Alibaba-NLP/Tongyi-DeepResearch-30B-A3B","name":"Tongyi DeepResearch 30B A3B","created":1758210804,"description":"Tongyi DeepResearch is an agentic large language model developed by Tongyi Lab, with 30 billion total parameters activating only 3 billion per token. It's optimized for long-horizon, deep information-seeking tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000045","input_cache_read":"0.00000009"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/alibaba/tongyi-deepresearch-30b-a3b/endpoints"}},{"id":"qwen/qwen3-coder-flash","canonical_slug":"qwen/qwen3-coder-flash","hugging_face_id":"","name":"Qwen: Qwen3 Coder Flash","created":1758115536,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000195","completion":"0.000000975","input_cache_read":"0.000000039","input_cache_write":"0.00000024375"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-flash/endpoints"}},{"id":"qwen/qwen3-next-80b-a3b-thinking","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","created":1757612284,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000000975","completion":"0.00000078"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-thinking-2509/endpoints"}},{"id":"qwen/qwen3-next-80b-a3b-instruct:free","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct (free)","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-instruct-2509/endpoints"}},{"id":"qwen/qwen3-next-80b-a3b-instruct","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.0000011"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-instruct-2509/endpoints"}},{"id":"qwen/qwen-plus-2025-07-28:thinking","canonical_slug":"qwen/qwen-plus-2025-07-28","hugging_face_id":"","name":"Qwen: Qwen Plus 0728 (thinking)","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000078","input_cache_write":"0.000000325"},"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-plus-2025-07-28/endpoints"}},{"id":"qwen/qwen-plus-2025-07-28","canonical_slug":"qwen/qwen-plus-2025-07-28","hugging_face_id":"","name":"Qwen: Qwen Plus 0728","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000078","input_cache_write":"0.000000325"},"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-plus-2025-07-28/endpoints"}},{"id":"nvidia/nemotron-nano-9b-v2:free","canonical_slug":"nvidia/nemotron-nano-9b-v2","hugging_face_id":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","name":"NVIDIA: Nemotron Nano 9B V2 (free)","created":1757106807,"description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-nano-9b-v2/endpoints"}},{"id":"nvidia/nemotron-nano-9b-v2","canonical_slug":"nvidia/nemotron-nano-9b-v2","hugging_face_id":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","name":"NVIDIA: Nemotron Nano 9B V2","created":1757106807,"description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000004","completion":"0.00000016"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-nano-9b-v2/endpoints"}},{"id":"moonshotai/kimi-k2-0905","canonical_slug":"moonshotai/kimi-k2-0905","hugging_face_id":"moonshotai/Kimi-K2-Instruct-0905","name":"MoonshotAI: Kimi K2 0905","created":1757021147,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000025"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2-0905/endpoints"}},{"id":"qwen/qwen3-30b-a3b-thinking-2507","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","hugging_face_id":"Qwen/Qwen3-30B-A3B-Thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","created":1756399192,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.0000004","input_cache_read":"0.00000008"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-thinking-2507/endpoints"}},{"id":"x-ai/grok-code-fast-1","canonical_slug":"x-ai/grok-code-fast-1","hugging_face_id":"","name":"xAI: Grok Code Fast 1","created":1756238927,"description":"Grok Code Fast 1 is a speedy and economical reasoning model that excels at agentic coding. With reasoning traces visible in the response, developers can steer Grok Code for high-quality...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000015","web_search":"0.005","input_cache_read":"0.00000002"},"top_provider":{"context_length":256000,"max_completion_tokens":10000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-code-fast-1/endpoints"}},{"id":"nousresearch/hermes-4-70b","canonical_slug":"nousresearch/hermes-4-70b","hugging_face_id":"NousResearch/Hermes-4-70B","name":"Nous: Hermes 4 70B","created":1756236182,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":"0.00000013","completion":"0.0000004"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-4-70b/endpoints"}},{"id":"nousresearch/hermes-4-405b","canonical_slug":"nousresearch/hermes-4-405b","hugging_face_id":"NousResearch/Hermes-4-405B","name":"Nous: Hermes 4 405B","created":1756235463,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000003"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-4-405b/endpoints"}},{"id":"deepseek/deepseek-chat-v3.1","canonical_slug":"deepseek/deepseek-chat-v3.1","hugging_face_id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.00000021","completion":"0.00000079","input_cache_read":"0.00000013"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3.1/endpoints"}},{"id":"openai/gpt-4o-audio-preview","canonical_slug":"openai/gpt-4o-audio-preview","hugging_face_id":"","name":"OpenAI: GPT-4o Audio","created":1755233061,"description":"The gpt-4o-audio-preview model adds support for audio inputs as prompts. This enhancement allows the model to detect nuances within audio recordings and add depth to generated user experiences. Audio outputs...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["audio","text"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","audio":"0.00004"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-audio-preview/endpoints"}},{"id":"mistralai/mistral-medium-3.1","canonical_slug":"mistralai/mistral-medium-3.1","hugging_face_id":"","name":"Mistral: Mistral Medium 3.1","created":1755095639,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-medium-3.1/endpoints"}},{"id":"baidu/ernie-4.5-21b-a3b","canonical_slug":"baidu/ernie-4.5-21b-a3b","hugging_face_id":"baidu/ERNIE-4.5-21B-A3B-PT","name":"Baidu: ERNIE 4.5 21B A3B","created":1755034167,"description":"A sophisticated text-based Mixture-of-Experts (MoE) model featuring 21B total parameters with 3B activated per token, delivering exceptional multimodal understanding and generation through heterogeneous MoE structures and modality-isolated routing. Supporting an...","context_length":120000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.00000028"},"top_provider":{"context_length":120000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.8,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/baidu/ernie-4.5-21b-a3b/endpoints"}},{"id":"baidu/ernie-4.5-vl-28b-a3b","canonical_slug":"baidu/ernie-4.5-vl-28b-a3b","hugging_face_id":"baidu/ERNIE-4.5-VL-28B-A3B-PT","name":"Baidu: ERNIE 4.5 VL 28B A3B","created":1755032836,"description":"A powerful multimodal Mixture-of-Experts chat model featuring 28B total parameters with 3B activated per token, delivering exceptional text and vision understanding through its innovative heterogeneous MoE structure with modality-isolated routing....","context_length":30000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.00000056"},"top_provider":{"context_length":30000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/baidu/ernie-4.5-vl-28b-a3b/endpoints"}},{"id":"z-ai/glm-4.5v","canonical_slug":"z-ai/glm-4.5v","hugging_face_id":"zai-org/GLM-4.5V","name":"Z.ai: GLM 4.5V","created":1754922288,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000018","input_cache_read":"0.00000011"},"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.5v/endpoints"}},{"id":"ai21/jamba-large-1.7","canonical_slug":"ai21/jamba-large-1.7","hugging_face_id":"ai21labs/AI21-Jamba-Large-1.7","name":"AI21: Jamba Large 1.7","created":1754669020,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000008"},"top_provider":{"context_length":256000,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/ai21/jamba-large-1.7/endpoints"}},{"id":"openai/gpt-5-chat","canonical_slug":"openai/gpt-5-chat-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Chat","created":1754587837,"description":"GPT-5 Chat is designed for advanced, natural, multimodal, and context-aware conversations for enterprise applications.","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","structured_outputs"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-chat-2025-08-07/endpoints"}},{"id":"openai/gpt-5","canonical_slug":"openai/gpt-5-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-2025-08-07/endpoints"}},{"id":"openai/gpt-5-mini","canonical_slug":"openai/gpt-5-mini-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Mini","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.000000025"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-05-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-mini-2025-08-07/endpoints"}},{"id":"openai/gpt-5-nano","canonical_slug":"openai/gpt-5-nano-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Nano","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000004","input_cache_read":"0.00000001"},"top_provider":{"context_length":400000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-05-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-nano-2025-08-07/endpoints"}},{"id":"openai/gpt-oss-120b:free","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b (free)","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","stop","temperature","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"}},{"id":"openai/gpt-oss-120b","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000039","completion":"0.00000018"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"}},{"id":"openai/gpt-oss-20b:free","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b (free)","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","stop","temperature","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"}},{"id":"openai/gpt-oss-20b","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000014"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"}},{"id":"anthropic/claude-opus-4.1","canonical_slug":"anthropic/claude-4.1-opus-20250805","hugging_face_id":"","name":"Anthropic: Claude Opus 4.1","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.000075","web_search":"0.01","input_cache_read":"0.0000015","input_cache_write":"0.00001875"},"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.1-opus-20250805/endpoints"}},{"id":"mistralai/codestral-2508","canonical_slug":"mistralai/codestral-2508","hugging_face_id":"","name":"Mistral: Codestral 2508","created":1754079630,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000003"},"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/codestral-2508/endpoints"}},{"id":"qwen/qwen3-coder-30b-a3b-instruct","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","hugging_face_id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","created":1753972379,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":160000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.00000027"},"top_provider":{"context_length":160000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-30b-a3b-instruct/endpoints"}},{"id":"qwen/qwen3-30b-a3b-instruct-2507","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","hugging_face_id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","created":1753806965,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.0000003"},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-instruct-2507/endpoints"}},{"id":"z-ai/glm-4.5","canonical_slug":"z-ai/glm-4.5","hugging_face_id":"zai-org/GLM-4.5","name":"Z.ai: GLM 4.5","created":1753471347,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.5/endpoints"}},{"id":"z-ai/glm-4.5-air:free","canonical_slug":"z-ai/glm-4.5-air","hugging_face_id":"zai-org/GLM-4.5-Air","name":"Z.ai: GLM 4.5 Air (free)","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":96000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.5-air/endpoints"}},{"id":"z-ai/glm-4.5-air","canonical_slug":"z-ai/glm-4.5-air","hugging_face_id":"zai-org/GLM-4.5-Air","name":"Z.ai: GLM 4.5 Air","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025"},"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.5-air/endpoints"}},{"id":"qwen/qwen3-235b-a22b-thinking-2507","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","hugging_face_id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","created":1753449557,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.0000001495","completion":"0.000001495"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-thinking-2507/endpoints"}},{"id":"z-ai/glm-4-32b","canonical_slug":"z-ai/glm-4-32b-0414","hugging_face_id":"","name":"Z.ai: GLM 4 32B ","created":1753376617,"description":"GLM 4 32B is a cost-effective foundation language model. It can efficiently perform complex tasks and has significantly enhanced capabilities in tool use, online search, and code-related intelligent tasks. It...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000001"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4-32b-0414/endpoints"}},{"id":"qwen/qwen3-coder:free","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","hugging_face_id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen: Qwen3 Coder 480B A35B (free)","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262000,"max_completion_tokens":262000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-480b-a35b-07-25/endpoints"}},{"id":"qwen/qwen3-coder","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","hugging_face_id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000022","completion":"0.0000018"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-480b-a35b-07-25/endpoints"}},{"id":"bytedance/ui-tars-1.5-7b","canonical_slug":"bytedance/ui-tars-1.5-7b","hugging_face_id":"ByteDance-Seed/UI-TARS-1.5-7B","name":"ByteDance: UI-TARS 7B ","created":1753205056,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000002","input_cache_read":"0.0000001"},"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/bytedance/ui-tars-1.5-7b/endpoints"}},{"id":"google/gemini-2.5-flash-lite","canonical_slug":"google/gemini-2.5-flash-lite","hugging_face_id":"","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","image":"0.0000001","audio":"0.0000003","web_search":"0.014","internal_reasoning":"0.0000004","input_cache_read":"0.00000001","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-flash-lite/endpoints"}},{"id":"qwen/qwen3-235b-a22b-2507","canonical_slug":"qwen/qwen3-235b-a22b-07-25","hugging_face_id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000071","completion":"0.0000001"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-07-25/endpoints"}},{"id":"switchpoint/router","canonical_slug":"switchpoint/router","hugging_face_id":"","name":"Switchpoint Router","created":1752272899,"description":"Switchpoint AI's router instantly analyzes your request and directs it to the optimal AI from an ever-evolving library. As the world of LLMs advances, our router gets smarter, ensuring you...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000085","completion":"0.0000034"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/switchpoint/router/endpoints"}},{"id":"moonshotai/kimi-k2","canonical_slug":"moonshotai/kimi-k2","hugging_face_id":"moonshotai/Kimi-K2-Instruct","name":"MoonshotAI: Kimi K2 0711","created":1752263252,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000057","completion":"0.0000023"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2/endpoints"}},{"id":"mistralai/devstral-medium","canonical_slug":"mistralai/devstral-medium-2507","hugging_face_id":"","name":"Mistral: Devstral Medium","created":1752161321,"description":"Devstral Medium is a high-performance code generation and agentic reasoning model developed jointly by Mistral AI and All Hands AI. Positioned as a step up from Devstral Small, it achieves...","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/devstral-medium-2507/endpoints"}},{"id":"mistralai/devstral-small","canonical_slug":"mistralai/devstral-small-2507","hugging_face_id":"mistralai/Devstral-Small-2507","name":"Mistral: Devstral Small 1.1","created":1752160751,"description":"Devstral Small 1.1 is a 24B parameter open-weight language model for software engineering agents, developed by Mistral AI in collaboration with All Hands AI. Finetuned from Mistral Small 3.1 and...","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003","input_cache_read":"0.00000001"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/devstral-small-2507/endpoints"}},{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition:free","canonical_slug":"venice/uncensored","hugging_face_id":"cognitivecomputations/Dolphin-Mistral-24B-Venice-Edition","name":"Venice: Uncensored (free)","created":1752094966,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/venice/uncensored/endpoints"}},{"id":"x-ai/grok-4","canonical_slug":"x-ai/grok-4-07-09","hugging_face_id":"","name":"xAI: Grok 4","created":1752087689,"description":"Grok 4 is xAI's latest reasoning model with a 256k context window. It supports parallel tool calling, structured outputs, and both image and text inputs. Note that reasoning is not...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005","input_cache_read":"0.00000075"},"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-07-31","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-4-07-09/endpoints"}},{"id":"tencent/hunyuan-a13b-instruct","canonical_slug":"tencent/hunyuan-a13b-instruct","hugging_face_id":"tencent/Hunyuan-A13B-Instruct","name":"Tencent: Hunyuan A13B Instruct","created":1751987664,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.00000057"},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/tencent/hunyuan-a13b-instruct/endpoints"}},{"id":"morph/morph-v3-large","canonical_slug":"morph/morph-v3-large","hugging_face_id":"","name":"Morph: Morph V3 Large","created":1751910858,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code}...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000009","completion":"0.0000019"},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/morph/morph-v3-large/endpoints"}},{"id":"morph/morph-v3-fast","canonical_slug":"morph/morph-v3-fast","hugging_face_id":"","name":"Morph: Morph V3 Fast","created":1751910002,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code} {edit_snippet}...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000012"},"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/morph/morph-v3-fast/endpoints"}},{"id":"baidu/ernie-4.5-vl-424b-a47b","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","hugging_face_id":"baidu/ERNIE-4.5-VL-424B-A47B-PT","name":"Baidu: ERNIE 4.5 VL 424B A47B ","created":1751300903,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":123000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000042","completion":"0.00000125"},"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/baidu/ernie-4.5-vl-424b-a47b/endpoints"}},{"id":"baidu/ernie-4.5-300b-a47b","canonical_slug":"baidu/ernie-4.5-300b-a47b","hugging_face_id":"baidu/ERNIE-4.5-300B-A47B-PT","name":"Baidu: ERNIE 4.5 300B A47B ","created":1751300139,"description":"ERNIE-4.5-300B-A47B is a 300B parameter Mixture-of-Experts (MoE) language model developed by Baidu as part of the ERNIE 4.5 series. It activates 47B parameters per token and supports text generation in...","context_length":123000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000028","completion":"0.0000011"},"top_provider":{"context_length":123000,"max_completion_tokens":12000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/baidu/ernie-4.5-300b-a47b/endpoints"}},{"id":"mistralai/mistral-small-3.2-24b-instruct","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","hugging_face_id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000002"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-3.2-24b-instruct-2506/endpoints"}},{"id":"minimax/minimax-m1","canonical_slug":"minimax/minimax-m1","hugging_face_id":"","name":"MiniMax: MiniMax M1","created":1750200414,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.0000022"},"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m1/endpoints"}},{"id":"google/gemini-2.5-flash","canonical_slug":"google/gemini-2.5-flash","hugging_face_id":"","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","audio":"0.000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-flash/endpoints"}},{"id":"google/gemini-2.5-pro","canonical_slug":"google/gemini-2.5-pro","hugging_face_id":"","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","image":"0.00000125","audio":"0.00000125","web_search":"0.014","internal_reasoning":"0.00001","input_cache_read":"0.000000125","input_cache_write":"0.000000375"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-pro/endpoints"}},{"id":"openai/o3-pro","canonical_slug":"openai/o3-pro-2025-06-10","hugging_face_id":"","name":"OpenAI: o3 Pro","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00002","completion":"0.00008","web_search":"0.01"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/o3-pro-2025-06-10/endpoints"}},{"id":"x-ai/grok-3-mini","canonical_slug":"x-ai/grok-3-mini","hugging_face_id":"","name":"xAI: Grok 3 Mini","created":1749583245,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000005","web_search":"0.005","input_cache_read":"0.000000075"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-02-28","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-3-mini/endpoints"}},{"id":"x-ai/grok-3","canonical_slug":"x-ai/grok-3","hugging_face_id":"","name":"xAI: Grok 3","created":1749582908,"description":"Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005","input_cache_read":"0.00000075"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-02-28","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-3/endpoints"}},{"id":"google/gemini-2.5-pro-preview","canonical_slug":"google/gemini-2.5-pro-preview-06-05","hugging_face_id":"","name":"Google: Gemini 2.5 Pro Preview 06-05","created":1749137257,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","image":"0.00000125","audio":"0.00000125","web_search":"0.014","internal_reasoning":"0.00001","input_cache_read":"0.000000125","input_cache_write":"0.000000375"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-pro-preview-06-05/endpoints"}},{"id":"deepseek/deepseek-r1-0528","canonical_slug":"deepseek/deepseek-r1-0528","hugging_face_id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-r1-0528/endpoints"}},{"id":"anthropic/claude-opus-4","canonical_slug":"anthropic/claude-4-opus-20250522","hugging_face_id":"","name":"Anthropic: Claude Opus 4","created":1747931245,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.000075","web_search":"0.01","input_cache_read":"0.0000015","input_cache_write":"0.00001875"},"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4-opus-20250522/endpoints"}},{"id":"anthropic/claude-sonnet-4","canonical_slug":"anthropic/claude-4-sonnet-20250522","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4","created":1747930371,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375"},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4-sonnet-20250522/endpoints"}},{"id":"google/gemma-3n-e4b-it","canonical_slug":"google/gemma-3n-e4b-it","hugging_face_id":"google/gemma-3n-E4B-it","name":"Google: Gemma 3n 4B","created":1747776824,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000012"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3n-e4b-it/endpoints"}},{"id":"mistralai/mistral-medium-3","canonical_slug":"mistralai/mistral-medium-3","hugging_face_id":"","name":"Mistral: Mistral Medium 3","created":1746627341,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-medium-3/endpoints"}},{"id":"google/gemini-2.5-pro-preview-05-06","canonical_slug":"google/gemini-2.5-pro-preview-03-25","hugging_face_id":"","name":"Google: Gemini 2.5 Pro Preview 05-06","created":1746578513,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","image":"0.00000125","audio":"0.00000125","web_search":"0.014","internal_reasoning":"0.00001","input_cache_read":"0.000000125","input_cache_write":"0.000000375"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-pro-preview-03-25/endpoints"}},{"id":"arcee-ai/spotlight","canonical_slug":"arcee-ai/spotlight","hugging_face_id":"","name":"Arcee AI: Spotlight","created":1746481552,"description":"Spotlight is a 7‑billion‑parameter vision‑language model derived from Qwen 2.5‑VL and fine‑tuned by Arcee AI for tight image‑text grounding tasks. It offers a 32 k‑token context window, enabling rich multimodal...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000018","completion":"0.00000018"},"top_provider":{"context_length":131072,"max_completion_tokens":65537,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/spotlight/endpoints"}},{"id":"arcee-ai/maestro-reasoning","canonical_slug":"arcee-ai/maestro-reasoning","hugging_face_id":"","name":"Arcee AI: Maestro Reasoning","created":1746481269,"description":"Maestro Reasoning is Arcee's flagship analysis model: a 32 B‑parameter derivative of Qwen 2.5‑32 B tuned with DPO and chain‑of‑thought RL for step‑by‑step logic. Compared to the earlier 7 B...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000009","completion":"0.0000033"},"top_provider":{"context_length":131072,"max_completion_tokens":32000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/maestro-reasoning/endpoints"}},{"id":"arcee-ai/virtuoso-large","canonical_slug":"arcee-ai/virtuoso-large","hugging_face_id":"","name":"Arcee AI: Virtuoso Large","created":1746478885,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000012"},"top_provider":{"context_length":131072,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/virtuoso-large/endpoints"}},{"id":"arcee-ai/coder-large","canonical_slug":"arcee-ai/coder-large","hugging_face_id":"","name":"Arcee AI: Coder Large","created":1746478663,"description":"Coder‑Large is a 32 B‑parameter offspring of Qwen 2.5‑Instruct that has been further trained on permissively‑licensed GitHub, CodeSearchNet and synthetic bug‑fix corpora. It supports a 32k context window, enabling multi‑file...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000008"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/arcee-ai/coder-large/endpoints"}},{"id":"meta-llama/llama-guard-4-12b","canonical_slug":"meta-llama/llama-guard-4-12b","hugging_face_id":"meta-llama/Llama-Guard-4-12B","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":163840,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000018","completion":"0.00000018"},"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-guard-4-12b/endpoints"}},{"id":"qwen/qwen3-30b-a3b","canonical_slug":"qwen/qwen3-30b-a3b-04-28","hugging_face_id":"Qwen/Qwen3-30B-A3B","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000009","completion":"0.00000045"},"top_provider":{"context_length":40960,"max_completion_tokens":20000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-04-28/endpoints"}},{"id":"qwen/qwen3-8b","canonical_slug":"qwen/qwen3-8b-04-28","hugging_face_id":"Qwen/Qwen3-8B","name":"Qwen: Qwen3 8B","created":1745876632,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000005","completion":"0.0000004","input_cache_read":"0.00000005"},"top_provider":{"context_length":40960,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-8b-04-28/endpoints"}},{"id":"qwen/qwen3-14b","canonical_slug":"qwen/qwen3-14b-04-28","hugging_face_id":"Qwen/Qwen3-14B","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.0000001","completion":"0.00000024"},"top_provider":{"context_length":40960,"max_completion_tokens":40960,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-14b-04-28/endpoints"}},{"id":"qwen/qwen3-32b","canonical_slug":"qwen/qwen3-32b-04-28","hugging_face_id":"Qwen/Qwen3-32B","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-32b-04-28/endpoints"}},{"id":"qwen/qwen3-235b-a22b","canonical_slug":"qwen/qwen3-235b-a22b-04-28","hugging_face_id":"Qwen/Qwen3-235B-A22B","name":"Qwen: Qwen3 235B A22B","created":1745875757,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.000000455","completion":"0.00000182"},"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-04-28/endpoints"}},{"id":"openai/o4-mini-high","canonical_slug":"openai/o4-mini-high-2025-04-16","hugging_face_id":"","name":"OpenAI: o4 Mini High","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/o4-mini-high-2025-04-16/endpoints"}},{"id":"openai/o3","canonical_slug":"openai/o3-2025-04-16","hugging_face_id":"","name":"OpenAI: o3","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/o3-2025-04-16/endpoints"}},{"id":"openai/o4-mini","canonical_slug":"openai/o4-mini-2025-04-16","hugging_face_id":"","name":"OpenAI: o4 Mini","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/o4-mini-2025-04-16/endpoints"}},{"id":"openai/gpt-4.1","canonical_slug":"openai/gpt-4.1-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000008","input_cache_read":"0.0000005"},"top_provider":{"context_length":1047576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4.1-2025-04-14/endpoints"}},{"id":"openai/gpt-4.1-mini","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 Mini","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4.1-mini-2025-04-14/endpoints"}},{"id":"openai/gpt-4.1-nano","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 Nano","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4.1-nano-2025-04-14/endpoints"}},{"id":"alfredpros/codellama-7b-instruct-solidity","canonical_slug":"alfredpros/codellama-7b-instruct-solidity","hugging_face_id":"AlfredPros/CodeLlama-7b-Instruct-Solidity","name":"AlfredPros: CodeLLaMa 7B Instruct Solidity","created":1744641874,"description":"A finetuned 7 billion parameters Code LLaMA - Instruct model to generate Solidity smart contract using 4-bit QLoRA finetuning provided by PEFT library.","context_length":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"alpaca"},"pricing":{"prompt":"0.0000008","completion":"0.0000012"},"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/api/v1/models/alfredpros/codellama-7b-instruct-solidity/endpoints"}},{"id":"x-ai/grok-3-mini-beta","canonical_slug":"x-ai/grok-3-mini-beta","hugging_face_id":"","name":"xAI: Grok 3 Mini Beta","created":1744240195,"description":"Grok 3 Mini is a lightweight, smaller thinking model. Unlike traditional models that generate answers immediately, Grok 3 Mini thinks before responding. It’s ideal for reasoning-heavy tasks that don’t demand...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000005","web_search":"0.005","input_cache_read":"0.000000075"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-02-28","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-3-mini-beta/endpoints"}},{"id":"x-ai/grok-3-beta","canonical_slug":"x-ai/grok-3-beta","hugging_face_id":"","name":"xAI: Grok 3 Beta","created":1744240068,"description":"Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005","input_cache_read":"0.00000075"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-02-28","expiration_date":"2026-05-15","links":{"details":"/api/v1/models/x-ai/grok-3-beta/endpoints"}},{"id":"meta-llama/llama-4-maverick","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","hugging_face_id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-4-maverick-17b-128e-instruct/endpoints"}},{"id":"meta-llama/llama-4-scout","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","hugging_face_id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":327680,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.0000003"},"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-4-scout-17b-16e-instruct/endpoints"}},{"id":"deepseek/deepseek-chat-v3-0324","canonical_slug":"deepseek/deepseek-chat-v3-0324","hugging_face_id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek: DeepSeek V3 0324","created":1742824755,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.00000077","input_cache_read":"0.000000135"},"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3-0324/endpoints"}},{"id":"openai/o1-pro","canonical_slug":"openai/o1-pro","hugging_face_id":"","name":"OpenAI: o1-pro","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00015","completion":"0.0006"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/o1-pro/endpoints"}},{"id":"mistralai/mistral-small-3.1-24b-instruct","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","hugging_face_id":"mistralai/Mistral-Small-3.1-24B-Instruct-2503","name":"Mistral: Mistral Small 3.1 24B","created":1742238937,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000035","completion":"0.00000056"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-3.1-24b-instruct-2503/endpoints"}},{"id":"google/gemma-3-4b-it","canonical_slug":"google/gemma-3-4b-it","hugging_face_id":"google/gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000004","completion":"0.00000008"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-4b-it/endpoints"}},{"id":"google/gemma-3-12b-it","canonical_slug":"google/gemma-3-12b-it","hugging_face_id":"google/gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000004","completion":"0.00000013"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-12b-it/endpoints"}},{"id":"cohere/command-a","canonical_slug":"cohere/command-a-03-2025","hugging_face_id":"CohereForAI/c4ai-command-a-03-2025","name":"Cohere: Command A","created":1741894342,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/cohere/command-a-03-2025/endpoints"}},{"id":"openai/gpt-4o-mini-search-preview","canonical_slug":"openai/gpt-4o-mini-search-preview-2025-03-11","hugging_face_id":"","name":"OpenAI: GPT-4o-mini Search Preview","created":1741818122,"description":"GPT-4o mini Search Preview is a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","web_search":"0.0275"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","structured_outputs","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-mini-search-preview-2025-03-11/endpoints"}},{"id":"openai/gpt-4o-search-preview","canonical_slug":"openai/gpt-4o-search-preview-2025-03-11","hugging_face_id":"","name":"OpenAI: GPT-4o Search Preview","created":1741817949,"description":"GPT-4o Search Previewis a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","web_search":"0.035"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","structured_outputs","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-search-preview-2025-03-11/endpoints"}},{"id":"rekaai/reka-flash-3","canonical_slug":"rekaai/reka-flash-3","hugging_face_id":"RekaAI/reka-flash-3","name":"Reka Flash 3","created":1741812813,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/rekaai/reka-flash-3/endpoints"}},{"id":"google/gemma-3-27b-it","canonical_slug":"google/gemma-3-27b-it","hugging_face_id":"google/gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000008","completion":"0.00000016"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-27b-it/endpoints"}},{"id":"thedrummer/skyfall-36b-v2","canonical_slug":"thedrummer/skyfall-36b-v2","hugging_face_id":"TheDrummer/Skyfall-36B-v2","name":"TheDrummer: Skyfall 36B V2","created":1741636566,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000008","input_cache_read":"0.00000025"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/thedrummer/skyfall-36b-v2/endpoints"}},{"id":"perplexity/sonar-reasoning-pro","canonical_slug":"perplexity/sonar-reasoning-pro","hugging_face_id":"","name":"Perplexity: Sonar Reasoning Pro","created":1741313308,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/perplexity/sonar-reasoning-pro/endpoints"}},{"id":"perplexity/sonar-pro","canonical_slug":"perplexity/sonar-pro","hugging_face_id":"","name":"Perplexity: Sonar Pro","created":1741312423,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005"},"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/perplexity/sonar-pro/endpoints"}},{"id":"perplexity/sonar-deep-research","canonical_slug":"perplexity/sonar-deep-research","hugging_face_id":"","name":"Perplexity: Sonar Deep Research","created":1741311246,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005","internal_reasoning":"0.000003"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/perplexity/sonar-deep-research/endpoints"}},{"id":"google/gemini-2.0-flash-lite-001","canonical_slug":"google/gemini-2.0-flash-lite-001","hugging_face_id":"","name":"Google: Gemini 2.0 Flash Lite","created":1740506212,"description":"Gemini 2.0 Flash Lite offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5),...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003","image":"0.000000075","audio":"0.000000075","web_search":"0.014","internal_reasoning":"0.0000003"},"top_provider":{"context_length":1048576,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":"2026-06-01","links":{"details":"/api/v1/models/google/gemini-2.0-flash-lite-001/endpoints"}},{"id":"mistralai/mistral-saba","canonical_slug":"mistralai/mistral-saba-2502","hugging_face_id":"","name":"Mistral: Saba","created":1739803239,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-saba-2502/endpoints"}},{"id":"meta-llama/llama-guard-3-8b","canonical_slug":"meta-llama/llama-guard-3-8b","hugging_face_id":"meta-llama/Llama-Guard-3-8B","name":"Llama Guard 3 8B","created":1739401318,"description":"Llama Guard 3 is a Llama-3.1-8B pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM inputs (prompt classification)...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"none"},"pricing":{"prompt":"0.00000048","completion":"0.00000003"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-guard-3-8b/endpoints"}},{"id":"openai/o3-mini-high","canonical_slug":"openai/o3-mini-high-2025-01-31","hugging_face_id":"","name":"OpenAI: o3 Mini High","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","input_cache_read":"0.00000055"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/o3-mini-high-2025-01-31/endpoints"}},{"id":"google/gemini-2.0-flash-001","canonical_slug":"google/gemini-2.0-flash-001","hugging_face_id":"","name":"Google: Gemini 2.0 Flash","created":1738769413,"description":"Gemini Flash 2.0 offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5). It...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","image":"0.0000001","audio":"0.0000007","web_search":"0.014","internal_reasoning":"0.0000004","input_cache_read":"0.000000025","input_cache_write":"0.00000008333333333333334"},"top_provider":{"context_length":1048576,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":"2026-06-01","links":{"details":"/api/v1/models/google/gemini-2.0-flash-001/endpoints"}},{"id":"aion-labs/aion-1.0","canonical_slug":"aion-labs/aion-1.0","hugging_face_id":"","name":"AionLabs: Aion-1.0","created":1738697557,"description":"Aion-1.0 is a multi-model system designed for high performance across various tasks, including reasoning and coding. It is built on DeepSeek-R1, augmented with additional models and techniques such as Tree...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000004","completion":"0.000008"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/aion-labs/aion-1.0/endpoints"}},{"id":"aion-labs/aion-1.0-mini","canonical_slug":"aion-labs/aion-1.0-mini","hugging_face_id":"FuseAI/FuseO1-DeepSeekR1-QwQ-SkyT1-32B-Preview","name":"AionLabs: Aion-1.0-Mini","created":1738697107,"description":"Aion-1.0-Mini 32B parameter model is a distilled version of the DeepSeek-R1 model, designed for strong performance in reasoning domains such as mathematics, coding, and logic. It is a modified variant...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000007","completion":"0.0000014"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/aion-labs/aion-1.0-mini/endpoints"}},{"id":"aion-labs/aion-rp-llama-3.1-8b","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","hugging_face_id":"","name":"AionLabs: Aion-RP 1.0 (8B)","created":1738696718,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000016"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/aion-labs/aion-rp-llama-3.1-8b/endpoints"}},{"id":"qwen/qwen2.5-vl-72b-instruct","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","hugging_face_id":"Qwen/Qwen2.5-VL-72B-Instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","created":1738410311,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":32000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.00000075"},"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen2.5-vl-72b-instruct/endpoints"}},{"id":"qwen/qwen-plus","canonical_slug":"qwen/qwen-plus-2025-01-25","hugging_face_id":"","name":"Qwen: Qwen-Plus","created":1738409840,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000078","input_cache_read":"0.000000052","input_cache_write":"0.000000325"},"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-plus-2025-01-25/endpoints"}},{"id":"openai/o3-mini","canonical_slug":"openai/o3-mini-2025-01-31","hugging_face_id":"","name":"OpenAI: o3 Mini","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","input_cache_read":"0.00000055"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/o3-mini-2025-01-31/endpoints"}},{"id":"mistralai/mistral-small-24b-instruct-2501","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","hugging_face_id":"mistralai/Mistral-Small-24B-Instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.00000008"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-24b-instruct-2501/endpoints"}},{"id":"deepseek/deepseek-r1-distill-qwen-32b","canonical_slug":"deepseek/deepseek-r1-distill-qwen-32b","hugging_face_id":"deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","name":"DeepSeek: R1 Distill Qwen 32B","created":1738194830,"description":"DeepSeek R1 Distill Qwen 32B is a distilled large language model based on [Qwen 2.5 32B](https://huggingface.co/Qwen/Qwen2.5-32B), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). It outperforms OpenAI's o1-mini across various benchmarks, achieving new...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.00000029","completion":"0.00000029"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-r1-distill-qwen-32b/endpoints"}},{"id":"perplexity/sonar","canonical_slug":"perplexity/sonar","hugging_face_id":"","name":"Perplexity: Sonar","created":1738013808,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000001","web_search":"0.005"},"top_provider":{"context_length":127072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/perplexity/sonar/endpoints"}},{"id":"deepseek/deepseek-r1-distill-llama-70b","canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","hugging_face_id":"deepseek-ai/DeepSeek-R1-Distill-Llama-70B","name":"DeepSeek: R1 Distill Llama 70B","created":1737663169,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.0000007","completion":"0.0000008"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-r1-distill-llama-70b/endpoints"}},{"id":"deepseek/deepseek-r1","canonical_slug":"deepseek/deepseek-r1","hugging_face_id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek: R1","created":1737381095,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":64000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.0000007","completion":"0.0000025"},"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_completion_tokens","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-r1/endpoints"}},{"id":"minimax/minimax-01","canonical_slug":"minimax/minimax-01","hugging_face_id":"MiniMaxAI/MiniMax-Text-01","name":"MiniMax: MiniMax-01","created":1736915462,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000011"},"top_provider":{"context_length":1000192,"max_completion_tokens":1000192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-01/endpoints"}},{"id":"microsoft/phi-4","canonical_slug":"microsoft/phi-4","hugging_face_id":"microsoft/phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000065","completion":"0.00000014"},"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/microsoft/phi-4/endpoints"}},{"id":"sao10k/l3.1-70b-hanami-x1","canonical_slug":"sao10k/l3.1-70b-hanami-x1","hugging_face_id":"Sao10K/L3.1-70B-Hanami-x1","name":"Sao10K: Llama 3.1 70B Hanami x1","created":1736302854,"description":"This is [Sao10K](/sao10k)'s experiment over [Euryale v2.2](/sao10k/l3.1-euryale-70b).","context_length":16000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000003"},"top_provider":{"context_length":16000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3.1-70b-hanami-x1/endpoints"}},{"id":"deepseek/deepseek-chat","canonical_slug":"deepseek/deepseek-chat-v3","hugging_face_id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.00000089"},"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3/endpoints"}},{"id":"sao10k/l3.3-euryale-70b","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","hugging_face_id":"Sao10K/L3.3-70B-Euryale-v2.3","name":"Sao10K: Llama 3.3 Euryale 70B","created":1734535928,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000065","completion":"0.00000075"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3.3-euryale-70b-v2.3/endpoints"}},{"id":"openai/o1","canonical_slug":"openai/o1-2024-12-17","hugging_face_id":"","name":"OpenAI: o1","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.00006","input_cache_read":"0.0000075"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/o1-2024-12-17/endpoints"}},{"id":"cohere/command-r7b-12-2024","canonical_slug":"cohere/command-r7b-12-2024","hugging_face_id":"","name":"Cohere: Command R7B (12-2024)","created":1734158152,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.0000000375","completion":"0.00000015"},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/cohere/command-r7b-12-2024/endpoints"}},{"id":"meta-llama/llama-3.3-70b-instruct:free","canonical_slug":"meta-llama/llama-3.3-70b-instruct","hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","name":"Meta: Llama 3.3 70B Instruct (free)","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.3-70b-instruct/endpoints"}},{"id":"meta-llama/llama-3.3-70b-instruct","canonical_slug":"meta-llama/llama-3.3-70b-instruct","hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.0000001","completion":"0.00000032"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.3-70b-instruct/endpoints"}},{"id":"amazon/nova-lite-v1","canonical_slug":"amazon/nova-lite-v1","hugging_face_id":"","name":"Amazon: Nova Lite 1.0","created":1733437363,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000024"},"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-lite-v1/endpoints"}},{"id":"amazon/nova-micro-v1","canonical_slug":"amazon/nova-micro-v1","hugging_face_id":"","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.000000035","completion":"0.00000014"},"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-micro-v1/endpoints"}},{"id":"amazon/nova-pro-v1","canonical_slug":"amazon/nova-pro-v1","hugging_face_id":"","name":"Amazon: Nova Pro 1.0","created":1733436303,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000032"},"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-pro-v1/endpoints"}},{"id":"openai/gpt-4o-2024-11-20","canonical_slug":"openai/gpt-4o-2024-11-20","hugging_face_id":"","name":"OpenAI: GPT-4o (2024-11-20)","created":1732127594,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-2024-11-20/endpoints"}},{"id":"mistralai/mistral-large-2411","canonical_slug":"mistralai/mistral-large-2411","hugging_face_id":"","name":"Mistral Large 2411","created":1731978685,"description":"Mistral Large 2 2411 is an update of [Mistral Large 2](/mistralai/mistral-large) released together with [Pixtral Large 2411](/mistralai/pixtral-large-2411) It provides a significant upgrade on the previous [Mistral Large 24.07](/mistralai/mistral-large-2407), with notable...","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-large-2411/endpoints"}},{"id":"mistralai/mistral-large-2407","canonical_slug":"mistralai/mistral-large-2407","hugging_face_id":"","name":"Mistral Large 2407","created":1731978415,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-large-2407/endpoints"}},{"id":"mistralai/pixtral-large-2411","canonical_slug":"mistralai/pixtral-large-2411","hugging_face_id":"","name":"Mistral: Pixtral Large 2411","created":1731977388,"description":"Pixtral Large is a 124B parameter, open-weight, multimodal model built on top of [Mistral Large 2](/mistralai/mistral-large-2411). The model is able to understand documents, charts and natural images. The model is...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/pixtral-large-2411/endpoints"}},{"id":"qwen/qwen-2.5-coder-32b-instruct","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","hugging_face_id":"Qwen/Qwen2.5-Coder-32B-Instruct","name":"Qwen2.5 Coder 32B Instruct","created":1731368400,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000066","completion":"0.000001"},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-2.5-coder-32b-instruct/endpoints"}},{"id":"thedrummer/unslopnemo-12b","canonical_slug":"thedrummer/unslopnemo-12b","hugging_face_id":"TheDrummer/UnslopNemo-12B-v4.1","name":"TheDrummer: UnslopNemo 12B","created":1731103448,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/thedrummer/unslopnemo-12b/endpoints"}},{"id":"anthropic/claude-3.5-haiku","canonical_slug":"anthropic/claude-3-5-haiku","hugging_face_id":null,"name":"Anthropic: Claude 3.5 Haiku","created":1730678400,"description":"Claude 3.5 Haiku features offers enhanced capabilities in speed, coding accuracy, and tool use. Engineered to excel in real-time applications, it delivers quick response times that are essential for dynamic...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.000004","web_search":"0.01","input_cache_read":"0.00000008","input_cache_write":"0.000001"},"top_provider":{"context_length":200000,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-3-5-haiku/endpoints"}},{"id":"anthracite-org/magnum-v4-72b","canonical_slug":"anthracite-org/magnum-v4-72b","hugging_face_id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","created":1729555200,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.000003","completion":"0.000005"},"top_provider":{"context_length":16384,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/anthracite-org/magnum-v4-72b/endpoints"}},{"id":"qwen/qwen-2.5-7b-instruct","canonical_slug":"qwen/qwen-2.5-7b-instruct","hugging_face_id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen: Qwen2.5 7B Instruct","created":1729036800,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000004","completion":"0.0000001"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-2.5-7b-instruct/endpoints"}},{"id":"inflection/inflection-3-productivity","canonical_slug":"inflection/inflection-3-productivity","hugging_face_id":null,"name":"Inflection: Inflection 3 Productivity","created":1728604800,"description":"Inflection 3 Productivity is optimized for following instructions. It is better for tasks requiring JSON output or precise adherence to provided guidelines. It has access to recent news. For emotional...","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":8000,"max_completion_tokens":1024,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/api/v1/models/inflection/inflection-3-productivity/endpoints"}},{"id":"inflection/inflection-3-pi","canonical_slug":"inflection/inflection-3-pi","hugging_face_id":null,"name":"Inflection: Inflection 3 Pi","created":1728604800,"description":"Inflection 3 Pi powers Inflection's [Pi](https://pi.ai) chatbot, including backstory, emotional intelligence, productivity, and safety. It has access to recent news, and excels in scenarios like customer support and roleplay. Pi...","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":8000,"max_completion_tokens":1024,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/api/v1/models/inflection/inflection-3-pi/endpoints"}},{"id":"thedrummer/rocinante-12b","canonical_slug":"thedrummer/rocinante-12b","hugging_face_id":"TheDrummer/Rocinante-12B-v1.1","name":"TheDrummer: Rocinante 12B","created":1727654400,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000017","completion":"0.00000043"},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/thedrummer/rocinante-12b/endpoints"}},{"id":"meta-llama/llama-3.2-1b-instruct","canonical_slug":"meta-llama/llama-3.2-1b-instruct","hugging_face_id":"meta-llama/Llama-3.2-1B-Instruct","name":"Meta: Llama 3.2 1B Instruct","created":1727222400,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.000000027","completion":"0.0000002"},"top_provider":{"context_length":60000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.2-1b-instruct/endpoints"}},{"id":"meta-llama/llama-3.2-3b-instruct:free","canonical_slug":"meta-llama/llama-3.2-3b-instruct","hugging_face_id":"meta-llama/Llama-3.2-3B-Instruct","name":"Meta: Llama 3.2 3B Instruct (free)","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.2-3b-instruct/endpoints"}},{"id":"meta-llama/llama-3.2-3b-instruct","canonical_slug":"meta-llama/llama-3.2-3b-instruct","hugging_face_id":"meta-llama/Llama-3.2-3B-Instruct","name":"Meta: Llama 3.2 3B Instruct","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":80000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.000000051","completion":"0.00000034"},"top_provider":{"context_length":80000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.2-3b-instruct/endpoints"}},{"id":"meta-llama/llama-3.2-11b-vision-instruct","canonical_slug":"meta-llama/llama-3.2-11b-vision-instruct","hugging_face_id":"meta-llama/Llama-3.2-11B-Vision-Instruct","name":"Meta: Llama 3.2 11B Vision Instruct","created":1727222400,"description":"Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.000000245","completion":"0.000000245"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.2-11b-vision-instruct/endpoints"}},{"id":"qwen/qwen-2.5-72b-instruct","canonical_slug":"qwen/qwen-2.5-72b-instruct","hugging_face_id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000036","completion":"0.0000004"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-2.5-72b-instruct/endpoints"}},{"id":"cohere/command-r-plus-08-2024","canonical_slug":"cohere/command-r-plus-08-2024","hugging_face_id":null,"name":"Cohere: Command R+ (08-2024)","created":1724976000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/api/v1/models/cohere/command-r-plus-08-2024/endpoints"}},{"id":"cohere/command-r-08-2024","canonical_slug":"cohere/command-r-08-2024","hugging_face_id":null,"name":"Cohere: Command R (08-2024)","created":1724976000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/api/v1/models/cohere/command-r-08-2024/endpoints"}},{"id":"sao10k/l3.1-euryale-70b","canonical_slug":"sao10k/l3.1-euryale-70b","hugging_face_id":"Sao10K/L3.1-70B-Euryale-v2.2","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000085","completion":"0.00000085"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3.1-euryale-70b/endpoints"}},{"id":"nousresearch/hermes-3-llama-3.1-70b","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-70B","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.0000003","completion":"0.0000003"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-70b/endpoints"}},{"id":"nousresearch/hermes-3-llama-3.1-405b:free","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-405B","name":"Nous: Hermes 3 405B Instruct (free)","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-405b/endpoints"}},{"id":"nousresearch/hermes-3-llama-3.1-405b","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-405B","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.000001","completion":"0.000001"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-405b/endpoints"}},{"id":"sao10k/l3-lunaris-8b","canonical_slug":"sao10k/l3-lunaris-8b","hugging_face_id":"Sao10K/L3-8B-Lunaris-v1","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"top_provider":{"context_length":8192,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3-lunaris-8b/endpoints"}},{"id":"openai/gpt-4o-2024-08-06","canonical_slug":"openai/gpt-4o-2024-08-06","hugging_face_id":null,"name":"OpenAI: GPT-4o (2024-08-06)","created":1722902400,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-2024-08-06/endpoints"}},{"id":"meta-llama/llama-3.1-70b-instruct","canonical_slug":"meta-llama/llama-3.1-70b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-70B-Instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.1-70b-instruct/endpoints"}},{"id":"meta-llama/llama-3.1-8b-instruct","canonical_slug":"meta-llama/llama-3.1-8b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000002","completion":"0.00000005"},"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.1-8b-instruct/endpoints"}},{"id":"mistralai/mistral-nemo","canonical_slug":"mistralai/mistral-nemo","hugging_face_id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":"0.00000002","completion":"0.00000003"},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-nemo/endpoints"}},{"id":"openai/gpt-4o-mini-2024-07-18","canonical_slug":"openai/gpt-4o-mini-2024-07-18","hugging_face_id":null,"name":"OpenAI: GPT-4o-mini (2024-07-18)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-mini-2024-07-18/endpoints"}},{"id":"openai/gpt-4o-mini","canonical_slug":"openai/gpt-4o-mini","hugging_face_id":null,"name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-mini/endpoints"}},{"id":"google/gemma-2-27b-it","canonical_slug":"google/gemma-2-27b-it","hugging_face_id":"google/gemma-2-27b-it","name":"Google: Gemma 2 27B","created":1720828800,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000065","completion":"0.00000065"},"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-2-27b-it/endpoints"}},{"id":"sao10k/l3-euryale-70b","canonical_slug":"sao10k/l3-euryale-70b","hugging_face_id":"Sao10K/L3-70B-Euryale-v2.1","name":"Sao10k: Llama 3 Euryale 70B v2.1","created":1718668800,"description":"Euryale 70B v2.1 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). - Better prompt adherence. - Better anatomy / spatial awareness. - Adapts much better to unique and custom...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000148","completion":"0.00000148"},"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3-euryale-70b/endpoints"}},{"id":"nousresearch/hermes-2-pro-llama-3-8b","canonical_slug":"nousresearch/hermes-2-pro-llama-3-8b","hugging_face_id":"NousResearch/Hermes-2-Pro-Llama-3-8B","name":"NousResearch: Hermes 2 Pro - Llama-3 8B","created":1716768000,"description":"Hermes 2 Pro is an upgraded, retrained version of Nous Hermes 2, consisting of an updated and cleaned version of the OpenHermes 2.5 Dataset, as well as a newly introduced...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.00000014","completion":"0.00000014"},"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-2-pro-llama-3-8b/endpoints"}},{"id":"openai/gpt-4o","canonical_slug":"openai/gpt-4o","hugging_face_id":null,"name":"OpenAI: GPT-4o","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o/endpoints"}},{"id":"openai/gpt-4o-2024-05-13","canonical_slug":"openai/gpt-4o-2024-05-13","hugging_face_id":null,"name":"OpenAI: GPT-4o (2024-05-13)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000015"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-2024-05-13/endpoints"}},{"id":"meta-llama/llama-3-8b-instruct","canonical_slug":"meta-llama/llama-3-8b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3-8B-Instruct","name":"Meta: Llama 3 8B Instruct","created":1713398400,"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 8B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000004","completion":"0.00000004"},"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3-8b-instruct/endpoints"}},{"id":"meta-llama/llama-3-70b-instruct","canonical_slug":"meta-llama/llama-3-70b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3-70B-Instruct","name":"Meta: Llama 3 70B Instruct","created":1713398400,"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 70B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000051","completion":"0.00000074"},"top_provider":{"context_length":8192,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3-70b-instruct/endpoints"}},{"id":"mistralai/mixtral-8x22b-instruct","canonical_slug":"mistralai/mixtral-8x22b-instruct","hugging_face_id":"mistralai/Mixtral-8x22B-Instruct-v0.1","name":"Mistral: Mixtral 8x22B Instruct","created":1713312000,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-01-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mixtral-8x22b-instruct/endpoints"}},{"id":"microsoft/wizardlm-2-8x22b","canonical_slug":"microsoft/wizardlm-2-8x22b","hugging_face_id":"microsoft/WizardLM-2-8x22B","name":"WizardLM-2 8x22B","created":1713225600,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65535,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":"0.00000062","completion":"0.00000062"},"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/microsoft/wizardlm-2-8x22b/endpoints"}},{"id":"openai/gpt-4-turbo","canonical_slug":"openai/gpt-4-turbo","hugging_face_id":null,"name":"OpenAI: GPT-4 Turbo","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00003"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4-turbo/endpoints"}},{"id":"anthropic/claude-3-haiku","canonical_slug":"anthropic/claude-3-haiku","hugging_face_id":null,"name":"Anthropic: Claude 3 Haiku","created":1710288000,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.00000125","input_cache_read":"0.00000003","input_cache_write":"0.0000003"},"top_provider":{"context_length":200000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-08-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-3-haiku/endpoints"}},{"id":"mistralai/mistral-large","canonical_slug":"mistralai/mistral-large","hugging_face_id":null,"name":"Mistral Large","created":1708905600,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-11-30","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-large/endpoints"}},{"id":"openai/gpt-3.5-turbo-0613","canonical_slug":"openai/gpt-3.5-turbo-0613","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo (older v0613)","created":1706140800,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000002"},"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-3.5-turbo-0613/endpoints"}},{"id":"openai/gpt-4-turbo-preview","canonical_slug":"openai/gpt-4-turbo-preview","hugging_face_id":null,"name":"OpenAI: GPT-4 Turbo Preview","created":1706140800,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00003"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4-turbo-preview/endpoints"}},{"id":"openrouter/auto","canonical_slug":"openrouter/auto","hugging_face_id":null,"name":"Auto Router","created":1699401600,"description":"Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...","context_length":2000000,"architecture":{"modality":"text+image+file+audio+video->text+image","input_modalities":["text","image","audio","file","video"],"output_modalities":["text","image"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"-1","completion":"-1"},"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p","web_search_options"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openrouter/auto/endpoints"}},{"id":"openai/gpt-4-1106-preview","canonical_slug":"openai/gpt-4-1106-preview","hugging_face_id":null,"name":"OpenAI: GPT-4 Turbo (older v1106)","created":1699228800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to April 2023.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00003"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-04-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4-1106-preview/endpoints"}},{"id":"mistralai/mistral-7b-instruct-v0.1","canonical_slug":"mistralai/mistral-7b-instruct-v0.1","hugging_face_id":"mistralai/Mistral-7B-Instruct-v0.1","name":"Mistral: Mistral 7B Instruct v0.1","created":1695859200,"description":"A 7.3B parameter model that outperforms Llama 2 13B on all benchmarks, with optimizations for speed and context length.","context_length":2824,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":"0.00000011","completion":"0.00000019"},"top_provider":{"context_length":2824,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2023-09-30","expiration_date":"2026-05-30","links":{"details":"/api/v1/models/mistralai/mistral-7b-instruct-v0.1/endpoints"}},{"id":"openai/gpt-3.5-turbo-instruct","canonical_slug":"openai/gpt-3.5-turbo-instruct","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo Instruct","created":1695859200,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":"0.0000015","completion":"0.000002"},"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-3.5-turbo-instruct/endpoints"}},{"id":"openai/gpt-3.5-turbo-16k","canonical_slug":"openai/gpt-3.5-turbo-16k","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo 16k","created":1693180800,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000004"},"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-3.5-turbo-16k/endpoints"}},{"id":"mancer/weaver","canonical_slug":"mancer/weaver","hugging_face_id":null,"name":"Mancer: Weaver (alpha)","created":1690934400,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.00000075","completion":"0.000001"},"top_provider":{"context_length":8000,"max_completion_tokens":2000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/api/v1/models/mancer/weaver/endpoints"}},{"id":"undi95/remm-slerp-l2-13b","canonical_slug":"undi95/remm-slerp-l2-13b","hugging_face_id":"Undi95/ReMM-SLERP-L2-13B","name":"ReMM SLERP 13B","created":1689984000,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.00000045","completion":"0.00000065"},"top_provider":{"context_length":6144,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/api/v1/models/undi95/remm-slerp-l2-13b/endpoints"}},{"id":"gryphe/mythomax-l2-13b","canonical_slug":"gryphe/mythomax-l2-13b","hugging_face_id":"Gryphe/MythoMax-L2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.00000006","completion":"0.00000006"},"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/api/v1/models/gryphe/mythomax-l2-13b/endpoints"}},{"id":"openai/gpt-4-0314","canonical_slug":"openai/gpt-4-0314","hugging_face_id":null,"name":"OpenAI: GPT-4 (older v0314)","created":1685232000,"description":"GPT-4-0314 is the first version of GPT-4 released, with a context length of 8,192 tokens, and was supported until June 14. Training data: up to Sep 2021.","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00006"},"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4-0314/endpoints"}},{"id":"openai/gpt-4","canonical_slug":"openai/gpt-4","hugging_face_id":null,"name":"OpenAI: GPT-4","created":1685232000,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00006"},"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4/endpoints"}},{"id":"openai/gpt-3.5-turbo","canonical_slug":"openai/gpt-3.5-turbo","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000015"},"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-3.5-turbo/endpoints"}}]} \ No newline at end of file diff --git a/package.json b/package.json index 44767672c..34a6a4979 100644 --- a/package.json +++ b/package.json @@ -24,7 +24,8 @@ "vercel:generate": "bun ./packages/core/script/generate-vercel.ts", "wandb:generate": "bun ./packages/core/script/generate-wandb.ts", "digitalocean:generate": "bun ./packages/core/script/generate-digitalocean.ts", - "ambient:generate": "bun ./packages/core/script/generate-ambient.ts" + "ambient:generate": "bun ./packages/core/script/generate-ambient.ts", + "openrouter:sync": "bun ./packages/core/script/sync-openrouter.ts" }, "dependencies": { "@cloudflare/workers-types": "^4.20260424.1", diff --git a/packages/core/script/sync-openrouter.ts b/packages/core/script/sync-openrouter.ts new file mode 100644 index 000000000..f39ea2588 --- /dev/null +++ b/packages/core/script/sync-openrouter.ts @@ -0,0 +1,309 @@ +#!/usr/bin/env bun + +import path from "node:path"; +import { mkdir, rm } from "node:fs/promises"; +import { z } from "zod"; + +import { ModelFamilyValues } from "../src/family.js"; +import { AuthoredModel, AuthoredModelShape } from "../src/schema.js"; + +const API_ENDPOINT = "https://openrouter.ai/api/v1/models"; + +const OpenRouterModel = z + .object({ + id: z.string(), + name: z.string(), + created: z.number(), + hugging_face_id: z.string().nullable(), + knowledge_cutoff: z.string().nullable(), + context_length: z.number(), + architecture: z.object({ + input_modalities: z.array(z.string()), + output_modalities: z.array(z.string()), + }), + pricing: z + .object({ + prompt: z.string(), + completion: z.string(), + internal_reasoning: z.string().optional(), + input_cache_read: z.string().optional(), + input_cache_write: z.string().optional(), + }), + top_provider: z.object({ + context_length: z.number().nullable(), + max_completion_tokens: z.number().nullable(), + }), + supported_parameters: z.array(z.string()), + }); + +const OpenRouterResponse = z + .object({ + data: z.array(OpenRouterModel), + }) + .passthrough(); + +const ExistingModel = AuthoredModelShape.partial() + .extend({ + extends: z + .object({ + from: z.string(), + omit: z.array(z.string()).optional(), + }) + .strict() + .optional(), + }) + .strict(); + +type OpenRouterModel = z.infer; +type ExistingModel = z.infer; + +function dateFromTimestamp(timestamp: number | undefined) { + if (timestamp === undefined) return new Date().toISOString().slice(0, 10); + return new Date(timestamp * 1000).toISOString().slice(0, 10); +} + +function formatInteger(n: number) { + return String(n).replace(/\B(?=(\d{3})+(?!\d))/g, "_"); +} + +function quote(value: string) { + return `"${value.replaceAll("\\", "\\\\").replaceAll('"', '\\"')}"`; +} + +function price(value: string | undefined) { + if (value === undefined) return undefined; + const number = Number(value); + return Number.isFinite(number) && number >= 0 + ? Math.round(number * 1_000_000_000_000) / 1_000_000 + : undefined; +} + +function modality(value: string) { + return value === "file" ? "pdf" : value; +} + +function modalities(values: string[] | undefined, fallback: string[]) { + const allowed = new Set(["text", "audio", "image", "video", "pdf"]); + const result = (values ?? fallback) + .map((value) => modality(value.toLowerCase())) + .filter((value) => allowed.has(value)); + return [...new Set(result.length > 0 ? result : fallback)]; +} + +function inferFamily(model: OpenRouterModel, name: string) { + const target = `${model.id} ${name}`.toLowerCase(); + return [...ModelFamilyValues] + .sort((a, b) => b.length - a.length) + .find((family) => target.includes(family.toLowerCase())); +} + +function buildModel(model: OpenRouterModel, existing: ExistingModel | undefined) { + const params = new Set(model.supported_parameters ?? []); + const name = model.name.replace(/^[^:]+:\s+/, ""); + const input = modalities(model.architecture?.input_modalities, ["text"]); + const output = modalities(model.architecture?.output_modalities, ["text"]); + const prompt = price(model.pricing?.prompt); + const completion = price(model.pricing?.completion); + const reasoning = params.has("reasoning") || params.has("include_reasoning"); + const context = model.top_provider?.context_length ?? model.context_length ?? 0; + const maxOutput = model.top_provider?.max_completion_tokens ?? existing?.limit?.output ?? context; + + return { + name, + family: existing?.family ?? inferFamily(model, name), + release_date: dateFromTimestamp(model.created), + last_updated: dateFromTimestamp(model.created), + attachment: input.some((value) => value !== "text"), + reasoning, + temperature: params.has("temperature"), + tool_call: params.has("tools") || params.has("tool_choice"), + structured_output: + params.has("structured_outputs") || params.has("response_format"), + knowledge: model.knowledge_cutoff?.slice(0, 10) ?? existing?.knowledge, + open_weights: Boolean(model.hugging_face_id), + status: existing?.status, + interleaved: existing?.interleaved, + cost: + prompt !== undefined && completion !== undefined + ? { + input: prompt, + output: completion, + reasoning: reasoning ? price(model.pricing?.internal_reasoning) : undefined, + cache_read: price(model.pricing?.input_cache_read), + cache_write: price(model.pricing?.input_cache_write), + tiers: existing?.cost?.tiers, + } + : existing?.cost, + limit: { + context, + input: existing?.limit?.input, + output: maxOutput, + }, + modalities: { input, output }, + }; +} + +function formatToml(model: ReturnType) { + const lines: string[] = []; + + lines.push(`name = ${quote(model.name)}`); + if (model.family) lines.push(`family = ${quote(model.family)}`); + lines.push(`release_date = ${quote(model.release_date)}`); + lines.push(`last_updated = ${quote(model.last_updated)}`); + lines.push(`attachment = ${model.attachment}`); + lines.push(`reasoning = ${model.reasoning}`); + lines.push(`temperature = ${model.temperature}`); + lines.push(`tool_call = ${model.tool_call}`); + lines.push(`structured_output = ${model.structured_output}`); + if (model.knowledge) lines.push(`knowledge = ${quote(model.knowledge)}`); + lines.push(`open_weights = ${model.open_weights}`); + if (model.status) lines.push(`status = ${quote(model.status)}`); + + if (model.interleaved !== undefined) { + lines.push(""); + if (model.interleaved === true) { + lines.push("interleaved = true"); + } else { + lines.push("[interleaved]"); + lines.push(`field = ${quote(model.interleaved.field)}`); + } + } + + if (model.cost) { + lines.push(""); + lines.push("[cost]"); + if (model.cost.input !== undefined) lines.push(`input = ${model.cost.input}`); + if (model.cost.output !== undefined) lines.push(`output = ${model.cost.output}`); + if (model.cost.reasoning !== undefined) { + lines.push(`reasoning = ${model.cost.reasoning}`); + } + if (model.cost.cache_read !== undefined) { + lines.push(`cache_read = ${model.cost.cache_read}`); + } + if (model.cost.cache_write !== undefined) { + lines.push(`cache_write = ${model.cost.cache_write}`); + } + + for (const tier of model.cost.tiers ?? []) { + lines.push(""); + lines.push("[[cost.tiers]]"); + lines.push(`tier = { size = ${formatInteger(tier.tier.size)} }`); + if (tier.input !== undefined) lines.push(`input = ${tier.input}`); + if (tier.output !== undefined) lines.push(`output = ${tier.output}`); + if (tier.cache_read !== undefined) lines.push(`cache_read = ${tier.cache_read}`); + if (tier.cache_write !== undefined) lines.push(`cache_write = ${tier.cache_write}`); + } + } + + lines.push(""); + lines.push("[limit]"); + lines.push(`context = ${formatInteger(model.limit.context)}`); + if (model.limit.input !== undefined) { + lines.push(`input = ${formatInteger(model.limit.input)}`); + } + lines.push(`output = ${formatInteger(model.limit.output)}`); + + lines.push(""); + lines.push("[modalities]"); + lines.push(`input = [${model.modalities.input.map(quote).join(", ")}]`); + lines.push(`output = [${model.modalities.output.map(quote).join(", ")}]`); + + return `${lines.join("\n")}\n`; +} + +async function main() { + const dryRun = process.argv.includes("--dry-run"); + const modelsDir = path.join( + import.meta.dirname, + "..", + "..", + "..", + "providers", + "openrouter", + "models", + ); + const headers = process.env.OPENROUTER_API_KEY + ? { Authorization: `Bearer ${process.env.OPENROUTER_API_KEY}` } + : undefined; + + const response = await fetch(API_ENDPOINT, { headers }); + if (!response.ok) { + throw new Error(`OpenRouter request failed: ${response.status} ${response.statusText}`); + } + + const parsed = OpenRouterResponse.safeParse(await response.json()); + if (!parsed.success) throw parsed.error; + + const existingFiles = new Set(); + for await (const file of new Bun.Glob("**/*.toml").scan({ cwd: modelsDir })) { + existingFiles.add(file); + } + + let created = 0; + let updated = 0; + let removed = 0; + let unchanged = 0; + const apiFiles = new Set(); + + for (const apiModel of parsed.data.data) { + const relativePath = `${apiModel.id}.toml`; + const filePath = path.join(modelsDir, relativePath); + const file = Bun.file(filePath); + const current = await file.exists() ? await file.text() : undefined; + const existing = current === undefined + ? undefined + : ExistingModel.parse(Bun.TOML.parse(current)); + const model = buildModel(apiModel, existing); + const next = formatToml(model); + + const valid = AuthoredModel.safeParse({ + id: relativePath.slice(0, -5), + ...Bun.TOML.parse(next), + }); + if (!valid.success) { + valid.error.cause = { relativePath }; + throw valid.error; + } + + apiFiles.add(relativePath); + + if (current === undefined) { + created++; + if (dryRun) { + console.log(`Would create ${relativePath}`); + } else { + await mkdir(path.dirname(filePath), { recursive: true }); + await Bun.write(filePath, next); + } + continue; + } + + if (current !== next) { + updated++; + if (dryRun) { + console.log(`Would update ${relativePath}`); + } else { + await Bun.write(filePath, next); + } + } else { + unchanged++; + } + } + + for (const relativePath of existingFiles) { + if (apiFiles.has(relativePath)) continue; + + removed++; + if (dryRun) { + console.log(`Would remove ${relativePath}`); + } else { + await rm(path.join(modelsDir, relativePath)); + } + } + + console.log( + `${dryRun ? "Dry run: " : ""}${created} created, ${updated} updated, ${removed} removed, ${unchanged} unchanged`, + ); +} + +await main(); diff --git a/providers/openrouter/models/ai21/jamba-large-1.7.toml b/providers/openrouter/models/ai21/jamba-large-1.7.toml new file mode 100644 index 000000000..d2e29964c --- /dev/null +++ b/providers/openrouter/models/ai21/jamba-large-1.7.toml @@ -0,0 +1,23 @@ +name = "Jamba Large 1.7" +family = "jamba" +release_date = "2025-08-08" +last_updated = "2025-08-08" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-08-31" +open_weights = true + +[cost] +input = 2 +output = 8 + +[limit] +context = 256_000 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/aion-labs/aion-1.0-mini.toml b/providers/openrouter/models/aion-labs/aion-1.0-mini.toml new file mode 100644 index 000000000..89437a3e1 --- /dev/null +++ b/providers/openrouter/models/aion-labs/aion-1.0-mini.toml @@ -0,0 +1,22 @@ +name = "Aion-1.0-Mini" +family = "o" +release_date = "2025-02-04" +last_updated = "2025-02-04" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = true + +[cost] +input = 0.7 +output = 1.4 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/aion-labs/aion-1.0.toml b/providers/openrouter/models/aion-labs/aion-1.0.toml new file mode 100644 index 000000000..ea36377af --- /dev/null +++ b/providers/openrouter/models/aion-labs/aion-1.0.toml @@ -0,0 +1,22 @@ +name = "Aion-1.0" +family = "o" +release_date = "2025-02-04" +last_updated = "2025-02-04" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 4 +output = 8 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/aion-labs/aion-2.0.toml b/providers/openrouter/models/aion-labs/aion-2.0.toml new file mode 100644 index 000000000..9ad424a84 --- /dev/null +++ b/providers/openrouter/models/aion-labs/aion-2.0.toml @@ -0,0 +1,23 @@ +name = "Aion-2.0" +family = "o" +release_date = "2026-02-23" +last_updated = "2026-02-23" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.8 +output = 1.6 +cache_read = 0.2 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/aion-labs/aion-rp-llama-3.1-8b.toml b/providers/openrouter/models/aion-labs/aion-rp-llama-3.1-8b.toml new file mode 100644 index 000000000..d1689c0a4 --- /dev/null +++ b/providers/openrouter/models/aion-labs/aion-rp-llama-3.1-8b.toml @@ -0,0 +1,23 @@ +name = "Aion-RP 1.0 (8B)" +family = "llama" +release_date = "2025-02-04" +last_updated = "2025-02-04" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = false + +[cost] +input = 0.8 +output = 1.6 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/alfredpros/codellama-7b-instruct-solidity.toml b/providers/openrouter/models/alfredpros/codellama-7b-instruct-solidity.toml new file mode 100644 index 000000000..b0e05c6e4 --- /dev/null +++ b/providers/openrouter/models/alfredpros/codellama-7b-instruct-solidity.toml @@ -0,0 +1,23 @@ +name = "CodeLLaMa 7B Instruct Solidity" +family = "llama" +release_date = "2025-04-14" +last_updated = "2025-04-14" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-06-30" +open_weights = true + +[cost] +input = 0.8 +output = 1.2 + +[limit] +context = 4_096 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/alibaba/tongyi-deepresearch-30b-a3b.toml b/providers/openrouter/models/alibaba/tongyi-deepresearch-30b-a3b.toml new file mode 100644 index 000000000..fab1ceb94 --- /dev/null +++ b/providers/openrouter/models/alibaba/tongyi-deepresearch-30b-a3b.toml @@ -0,0 +1,24 @@ +name = "Tongyi DeepResearch 30B A3B" +family = "yi" +release_date = "2025-09-18" +last_updated = "2025-09-18" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.09 +output = 0.45 +cache_read = 0.09 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/allenai/olmo-3-32b-think.toml b/providers/openrouter/models/allenai/olmo-3-32b-think.toml new file mode 100644 index 000000000..030319cfd --- /dev/null +++ b/providers/openrouter/models/allenai/olmo-3-32b-think.toml @@ -0,0 +1,22 @@ +name = "Olmo 3 32B Think" +family = "allenai" +release_date = "2025-11-21" +last_updated = "2025-11-21" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = true + +[cost] +input = 0.15 +output = 0.5 + +[limit] +context = 65_536 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/amazon/nova-2-lite-v1.toml b/providers/openrouter/models/amazon/nova-2-lite-v1.toml new file mode 100644 index 000000000..cb461ec94 --- /dev/null +++ b/providers/openrouter/models/amazon/nova-2-lite-v1.toml @@ -0,0 +1,22 @@ +name = "Nova 2 Lite" +family = "nova" +release_date = "2025-12-02" +last_updated = "2025-12-02" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[cost] +input = 0.3 +output = 2.5 + +[limit] +context = 1_000_000 +output = 65_535 + +[modalities] +input = ["text", "image", "video", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/amazon/nova-lite-v1.toml b/providers/openrouter/models/amazon/nova-lite-v1.toml new file mode 100644 index 000000000..45af1fd68 --- /dev/null +++ b/providers/openrouter/models/amazon/nova-lite-v1.toml @@ -0,0 +1,23 @@ +name = "Nova Lite 1.0" +family = "nova-lite" +release_date = "2024-12-05" +last_updated = "2024-12-05" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2024-10-31" +open_weights = false + +[cost] +input = 0.06 +output = 0.24 + +[limit] +context = 300_000 +output = 5_120 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/amazon/nova-micro-v1.toml b/providers/openrouter/models/amazon/nova-micro-v1.toml new file mode 100644 index 000000000..c57a76f4c --- /dev/null +++ b/providers/openrouter/models/amazon/nova-micro-v1.toml @@ -0,0 +1,23 @@ +name = "Nova Micro 1.0" +family = "nova-micro" +release_date = "2024-12-05" +last_updated = "2024-12-05" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2024-10-31" +open_weights = false + +[cost] +input = 0.035 +output = 0.14 + +[limit] +context = 128_000 +output = 5_120 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/amazon/nova-premier-v1.toml b/providers/openrouter/models/amazon/nova-premier-v1.toml new file mode 100644 index 000000000..a238e3389 --- /dev/null +++ b/providers/openrouter/models/amazon/nova-premier-v1.toml @@ -0,0 +1,23 @@ +name = "Nova Premier 1.0" +family = "nova" +release_date = "2025-10-31" +last_updated = "2025-10-31" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[cost] +input = 2.5 +output = 12.5 +cache_read = 0.625 + +[limit] +context = 1_000_000 +output = 32_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/amazon/nova-pro-v1.toml b/providers/openrouter/models/amazon/nova-pro-v1.toml new file mode 100644 index 000000000..a72cb7c1f --- /dev/null +++ b/providers/openrouter/models/amazon/nova-pro-v1.toml @@ -0,0 +1,23 @@ +name = "Nova Pro 1.0" +family = "nova-pro" +release_date = "2024-12-05" +last_updated = "2024-12-05" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2024-10-31" +open_weights = false + +[cost] +input = 0.8 +output = 3.2 + +[limit] +context = 300_000 +output = 5_120 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/anthracite-org/magnum-v4-72b.toml b/providers/openrouter/models/anthracite-org/magnum-v4-72b.toml new file mode 100644 index 000000000..d93b764ab --- /dev/null +++ b/providers/openrouter/models/anthracite-org/magnum-v4-72b.toml @@ -0,0 +1,23 @@ +name = "Magnum v4 72B" +family = "o" +release_date = "2024-10-22" +last_updated = "2024-10-22" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2024-06-30" +open_weights = true + +[cost] +input = 3 +output = 5 + +[limit] +context = 16_384 +output = 2_048 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-3-haiku.toml b/providers/openrouter/models/anthropic/claude-3-haiku.toml new file mode 100644 index 000000000..428127e6b --- /dev/null +++ b/providers/openrouter/models/anthropic/claude-3-haiku.toml @@ -0,0 +1,25 @@ +name = "Claude 3 Haiku" +family = "claude" +release_date = "2024-03-13" +last_updated = "2024-03-13" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2023-08-31" +open_weights = false + +[cost] +input = 0.25 +output = 1.25 +cache_read = 0.03 +cache_write = 0.3 + +[limit] +context = 200_000 +output = 4_096 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-3.5-haiku.toml b/providers/openrouter/models/anthropic/claude-3.5-haiku.toml index 4ee787370..9030b0f06 100644 --- a/providers/openrouter/models/anthropic/claude-3.5-haiku.toml +++ b/providers/openrouter/models/anthropic/claude-3.5-haiku.toml @@ -1,2 +1,25 @@ -[extends] -from = "anthropic/claude-3-5-haiku-20241022" +name = "Claude 3.5 Haiku" +family = "claude" +release_date = "2024-11-04" +last_updated = "2024-11-04" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2024-07-31" +open_weights = false + +[cost] +input = 0.8 +output = 4 +cache_read = 0.08 +cache_write = 1 + +[limit] +context = 200_000 +output = 8_192 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-haiku-4.5.toml b/providers/openrouter/models/anthropic/claude-haiku-4.5.toml index a3e42c86c..658ffbc10 100644 --- a/providers/openrouter/models/anthropic/claude-haiku-4.5.toml +++ b/providers/openrouter/models/anthropic/claude-haiku-4.5.toml @@ -11,9 +11,9 @@ knowledge = "2025-02-28" open_weights = false [cost] -input = 1.00 -output = 5.00 -cache_read = 0.10 +input = 1 +output = 5 +cache_read = 0.1 cache_write = 1.25 [limit] diff --git a/providers/openrouter/models/anthropic/claude-opus-4.1.toml b/providers/openrouter/models/anthropic/claude-opus-4.1.toml index 61c26560d..653f43ca4 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.1.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.1.toml @@ -7,13 +7,13 @@ reasoning = true temperature = true tool_call = true structured_output = true -knowledge = "2025-03-31" +knowledge = "2025-01-31" open_weights = false [cost] -input = 15.00 -output = 75.00 -cache_read = 1.50 +input = 15 +output = 75 +cache_read = 1.5 cache_write = 18.75 [limit] @@ -21,5 +21,5 @@ context = 200_000 output = 32_000 [modalities] -input = ["text", "image", "pdf"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-opus-4.5.toml b/providers/openrouter/models/anthropic/claude-opus-4.5.toml index a2eb8b5d0..dbc7a5b30 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.5.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.5.toml @@ -11,15 +11,15 @@ knowledge = "2025-05-30" open_weights = false [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [limit] context = 200_000 -output = 32_000 +output = 64_000 [modalities] -input = ["text", "image", "pdf"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml b/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml new file mode 100644 index 000000000..3b00d661b --- /dev/null +++ b/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml @@ -0,0 +1,24 @@ +name = "Claude Opus 4.6 (Fast)" +family = "claude-opus" +release_date = "2026-04-07" +last_updated = "2026-04-07" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 30 +output = 150 +cache_read = 3 +cache_write = 37.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-opus-4.6.toml b/providers/openrouter/models/anthropic/claude-opus-4.6.toml index 2867ccd55..0d9a1e4fd 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.6.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.6.toml @@ -1,7 +1,7 @@ name = "Claude Opus 4.6" family = "claude-opus" -release_date = "2026-02-05" -last_updated = "2026-02-05" +release_date = "2026-02-04" +last_updated = "2026-02-04" attachment = true reasoning = true temperature = true @@ -11,17 +11,17 @@ knowledge = "2025-05-31" open_weights = false [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] tier = { size = 200_000 } -input = 10.00 -output = 37.50 -cache_read = 1.00 -cache_write = 12.50 +input = 10 +output = 37.5 +cache_read = 1 +cache_write = 12.5 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml b/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml new file mode 100644 index 000000000..fd1df1a9c --- /dev/null +++ b/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml @@ -0,0 +1,24 @@ +name = "Claude Opus 4.7 (Fast)" +family = "claude-opus" +release_date = "2026-05-12" +last_updated = "2026-05-12" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 30 +output = 150 +cache_read = 3 +cache_write = 37.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-opus-4.7.toml b/providers/openrouter/models/anthropic/claude-opus-4.7.toml index 1e1e085b1..01e8ab0e8 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.7.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.7.toml @@ -11,17 +11,17 @@ knowledge = "2026-01-31" open_weights = false [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] tier = { size = 200_000 } -input = 10.00 -output = 37.50 -cache_read = 1.00 -cache_write = 12.50 +input = 10 +output = 37.5 +cache_read = 1 +cache_write = 12.5 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/anthropic/claude-opus-4.toml b/providers/openrouter/models/anthropic/claude-opus-4.toml index 4fd5a331e..1d12285db 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.toml @@ -1,2 +1,25 @@ -[extends] -from = "anthropic/claude-opus-4-20250514" +name = "Claude Opus 4" +family = "claude-opus" +release_date = "2025-05-22" +last_updated = "2025-05-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +knowledge = "2025-01-31" +open_weights = false + +[cost] +input = 15 +output = 75 +cache_read = 1.5 +cache_write = 18.75 + +[limit] +context = 200_000 +output = 32_000 + +[modalities] +input = ["image", "text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-sonnet-4.5.toml b/providers/openrouter/models/anthropic/claude-sonnet-4.5.toml index 267305c97..6206dc4ff 100644 --- a/providers/openrouter/models/anthropic/claude-sonnet-4.5.toml +++ b/providers/openrouter/models/anthropic/claude-sonnet-4.5.toml @@ -7,21 +7,21 @@ reasoning = true temperature = true tool_call = true structured_output = true -knowledge = "2025-07-31" +knowledge = "2025-01-31" open_weights = false [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 [[cost.tiers]] tier = { size = 200_000 } -input = 6.00 -output = 22.50 -cache_read = 0.60 -cache_write = 7.50 +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/anthropic/claude-sonnet-4.6.toml b/providers/openrouter/models/anthropic/claude-sonnet-4.6.toml index 45e9c810c..47d0b61bc 100644 --- a/providers/openrouter/models/anthropic/claude-sonnet-4.6.toml +++ b/providers/openrouter/models/anthropic/claude-sonnet-4.6.toml @@ -11,22 +11,22 @@ knowledge = "2025-08-31" open_weights = false [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 [[cost.tiers]] tier = { size = 200_000 } -input = 6.00 -output = 22.50 -cache_read = 0.60 -cache_write = 7.50 +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 [limit] context = 1_000_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-sonnet-4.toml b/providers/openrouter/models/anthropic/claude-sonnet-4.toml index ed949f4bb..323c6d971 100644 --- a/providers/openrouter/models/anthropic/claude-sonnet-4.toml +++ b/providers/openrouter/models/anthropic/claude-sonnet-4.toml @@ -6,26 +6,27 @@ attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2025-03-31" +structured_output = false +knowledge = "2025-01-31" open_weights = false [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 [[cost.tiers]] tier = { size = 200_000 } -input = 6.00 -output = 22.50 -cache_read = 0.60 -cache_write = 7.50 +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 [limit] -context = 200_000 +context = 1_000_000 output = 64_000 [modalities] -input = ["text", "image", "pdf"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/arcee-ai/coder-large.toml b/providers/openrouter/models/arcee-ai/coder-large.toml new file mode 100644 index 000000000..dfcacc8a3 --- /dev/null +++ b/providers/openrouter/models/arcee-ai/coder-large.toml @@ -0,0 +1,23 @@ +name = "Coder Large" +family = "o" +release_date = "2025-05-05" +last_updated = "2025-05-05" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-03-31" +open_weights = false + +[cost] +input = 0.5 +output = 0.8 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/arcee-ai/maestro-reasoning.toml b/providers/openrouter/models/arcee-ai/maestro-reasoning.toml new file mode 100644 index 000000000..43cd0d1be --- /dev/null +++ b/providers/openrouter/models/arcee-ai/maestro-reasoning.toml @@ -0,0 +1,23 @@ +name = "Maestro Reasoning" +family = "o" +release_date = "2025-05-05" +last_updated = "2025-05-05" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-03-31" +open_weights = false + +[cost] +input = 0.9 +output = 3.3 + +[limit] +context = 131_072 +output = 32_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/arcee-ai/spotlight.toml b/providers/openrouter/models/arcee-ai/spotlight.toml new file mode 100644 index 000000000..12ae908dc --- /dev/null +++ b/providers/openrouter/models/arcee-ai/spotlight.toml @@ -0,0 +1,23 @@ +name = "Spotlight" +family = "o" +release_date = "2025-05-05" +last_updated = "2025-05-05" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-03-31" +open_weights = false + +[cost] +input = 0.18 +output = 0.18 + +[limit] +context = 131_072 +output = 65_537 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/arcee-ai/trinity-large-preview:free.toml b/providers/openrouter/models/arcee-ai/trinity-large-preview.toml similarity index 59% rename from providers/openrouter/models/arcee-ai/trinity-large-preview:free.toml rename to providers/openrouter/models/arcee-ai/trinity-large-preview.toml index a6f7a2b96..1bb0e9844 100644 --- a/providers/openrouter/models/arcee-ai/trinity-large-preview:free.toml +++ b/providers/openrouter/models/arcee-ai/trinity-large-preview.toml @@ -1,23 +1,21 @@ name = "Trinity Large Preview" family = "trinity" -release_date = "2026-01-28" -last_updated = "2026-01-28" +release_date = "2026-01-27" +last_updated = "2026-01-27" attachment = false reasoning = false temperature = true -# may be inaccurate -knowledge = "2025-06" tool_call = true structured_output = true open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0.15 +output = 0.45 [limit] -context = 131_072 -output = 131_072 +context = 131_000 +output = 131_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/arcee-ai/trinity-large-thinking.toml b/providers/openrouter/models/arcee-ai/trinity-large-thinking.toml index db2c524eb..5bbce45dc 100644 --- a/providers/openrouter/models/arcee-ai/trinity-large-thinking.toml +++ b/providers/openrouter/models/arcee-ai/trinity-large-thinking.toml @@ -1,20 +1,22 @@ name = "Trinity Large Thinking" family = "trinity" +release_date = "2026-04-01" +last_updated = "2026-04-01" attachment = false reasoning = true -tool_call = true temperature = true -release_date = "2026-04-01" -last_updated = "2026-04-03" +tool_call = true +structured_output = true open_weights = true [cost] input = 0.22 output = 0.85 +cache_read = 0.06 [limit] context = 262_144 -output = 80_000 +output = 262_144 [modalities] input = ["text"] diff --git a/providers/openrouter/models/arcee-ai/trinity-large-thinking:free.toml b/providers/openrouter/models/arcee-ai/trinity-large-thinking:free.toml new file mode 100644 index 000000000..1e2060b89 --- /dev/null +++ b/providers/openrouter/models/arcee-ai/trinity-large-thinking:free.toml @@ -0,0 +1,22 @@ +name = "Trinity Large Thinking (free)" +family = "trinity" +release_date = "2026-04-01" +last_updated = "2026-04-01" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = true + +[cost] +input = 0 +output = 0 + +[limit] +context = 262_144 +output = 80_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-oss-120b:exacto.toml b/providers/openrouter/models/arcee-ai/trinity-mini.toml similarity index 57% rename from providers/openrouter/models/openai/gpt-oss-120b:exacto.toml rename to providers/openrouter/models/arcee-ai/trinity-mini.toml index 14a661700..6e6224c14 100644 --- a/providers/openrouter/models/openai/gpt-oss-120b:exacto.toml +++ b/providers/openrouter/models/arcee-ai/trinity-mini.toml @@ -1,7 +1,7 @@ -name = "GPT OSS 120B (exacto)" -family = "gpt-oss" -release_date = "2025-08-05" -last_updated = "2025-08-05" +name = "Trinity Mini" +family = "trinity-mini" +release_date = "2025-12-01" +last_updated = "2025-12-01" attachment = false reasoning = true temperature = true @@ -10,12 +10,12 @@ structured_output = true open_weights = true [cost] -input = 0.05 -output = 0.24 +input = 0.045 +output = 0.15 [limit] context = 131_072 -output = 32_768 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/arcee-ai/virtuoso-large.toml b/providers/openrouter/models/arcee-ai/virtuoso-large.toml new file mode 100644 index 000000000..4a30e9034 --- /dev/null +++ b/providers/openrouter/models/arcee-ai/virtuoso-large.toml @@ -0,0 +1,23 @@ +name = "Virtuoso Large" +family = "o" +release_date = "2025-05-05" +last_updated = "2025-05-05" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2025-03-31" +open_weights = false + +[cost] +input = 0.75 +output = 1.2 + +[limit] +context = 131_072 +output = 64_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/cobuddy:free.toml b/providers/openrouter/models/baidu/cobuddy:free.toml new file mode 100644 index 000000000..24040b7e2 --- /dev/null +++ b/providers/openrouter/models/baidu/cobuddy:free.toml @@ -0,0 +1,22 @@ +name = "CoBuddy (free)" +family = "o" +release_date = "2026-05-06" +last_updated = "2026-05-06" +attachment = false +reasoning = true +temperature = false +tool_call = true +structured_output = false +open_weights = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 131_072 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/ernie-4.5-21b-a3b-thinking.toml b/providers/openrouter/models/baidu/ernie-4.5-21b-a3b-thinking.toml new file mode 100644 index 000000000..16619cc79 --- /dev/null +++ b/providers/openrouter/models/baidu/ernie-4.5-21b-a3b-thinking.toml @@ -0,0 +1,23 @@ +name = "ERNIE 4.5 21B A3B Thinking" +family = "ernie" +release_date = "2025-10-09" +last_updated = "2025-10-09" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.07 +output = 0.28 + +[limit] +context = 131_072 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/ernie-4.5-21b-a3b.toml b/providers/openrouter/models/baidu/ernie-4.5-21b-a3b.toml new file mode 100644 index 000000000..1b3d83e32 --- /dev/null +++ b/providers/openrouter/models/baidu/ernie-4.5-21b-a3b.toml @@ -0,0 +1,23 @@ +name = "ERNIE 4.5 21B A3B" +family = "ernie" +release_date = "2025-08-12" +last_updated = "2025-08-12" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.07 +output = 0.28 + +[limit] +context = 120_000 +output = 8_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/ernie-4.5-300b-a47b.toml b/providers/openrouter/models/baidu/ernie-4.5-300b-a47b.toml new file mode 100644 index 000000000..951dbbda9 --- /dev/null +++ b/providers/openrouter/models/baidu/ernie-4.5-300b-a47b.toml @@ -0,0 +1,23 @@ +name = "ERNIE 4.5 300B A47B " +family = "ernie" +release_date = "2025-06-30" +last_updated = "2025-06-30" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.28 +output = 1.1 + +[limit] +context = 123_000 +output = 12_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/ernie-4.5-vl-28b-a3b.toml b/providers/openrouter/models/baidu/ernie-4.5-vl-28b-a3b.toml new file mode 100644 index 000000000..ce92631b1 --- /dev/null +++ b/providers/openrouter/models/baidu/ernie-4.5-vl-28b-a3b.toml @@ -0,0 +1,23 @@ +name = "ERNIE 4.5 VL 28B A3B" +family = "ernie" +release_date = "2025-08-12" +last_updated = "2025-08-12" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.14 +output = 0.56 + +[limit] +context = 30_000 +output = 8_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/ernie-4.5-vl-424b-a47b.toml b/providers/openrouter/models/baidu/ernie-4.5-vl-424b-a47b.toml new file mode 100644 index 000000000..b4688fced --- /dev/null +++ b/providers/openrouter/models/baidu/ernie-4.5-vl-424b-a47b.toml @@ -0,0 +1,23 @@ +name = "ERNIE 4.5 VL 424B A47B " +family = "ernie" +release_date = "2025-06-30" +last_updated = "2025-06-30" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.42 +output = 1.25 + +[limit] +context = 123_000 +output = 16_000 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/baidu/qianfan-ocr-fast.toml b/providers/openrouter/models/baidu/qianfan-ocr-fast.toml new file mode 100644 index 000000000..21ebb260a --- /dev/null +++ b/providers/openrouter/models/baidu/qianfan-ocr-fast.toml @@ -0,0 +1,22 @@ +name = "Qianfan-OCR-Fast" +family = "o" +release_date = "2026-04-20" +last_updated = "2026-04-20" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.68 +output = 2.81 + +[limit] +context = 65_536 +output = 28_672 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/black-forest-labs/flux.2-flex.toml b/providers/openrouter/models/black-forest-labs/flux.2-flex.toml deleted file mode 100644 index ff23af2f5..000000000 --- a/providers/openrouter/models/black-forest-labs/flux.2-flex.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "FLUX.2 Flex" -family = "flux" -release_date = "2025-11-25" -last_updated = "2026-01-31" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = false - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 67_344 -output = 67_344 - -[modalities] -input = ["image","text"] -output = ["image"] diff --git a/providers/openrouter/models/black-forest-labs/flux.2-klein-4b.toml b/providers/openrouter/models/black-forest-labs/flux.2-klein-4b.toml deleted file mode 100644 index 833b31694..000000000 --- a/providers/openrouter/models/black-forest-labs/flux.2-klein-4b.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "FLUX.2 Klein 4B" -family = "flux" -release_date = "2026-01-14" -last_updated = "2026-01-31" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 40_960 -output = 40_960 - -[modalities] -input = ["image","text"] -output = ["image"] diff --git a/providers/openrouter/models/black-forest-labs/flux.2-max.toml b/providers/openrouter/models/black-forest-labs/flux.2-max.toml deleted file mode 100644 index 669c71f6e..000000000 --- a/providers/openrouter/models/black-forest-labs/flux.2-max.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "FLUX.2 Max" -family = "flux" -release_date = "2025-12-16" -last_updated = "2026-01-31" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = false - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 46_864 -output = 46_864 - -[modalities] -input = ["image","text"] -output = ["image"] diff --git a/providers/openrouter/models/black-forest-labs/flux.2-pro.toml b/providers/openrouter/models/black-forest-labs/flux.2-pro.toml deleted file mode 100644 index 4b83ce031..000000000 --- a/providers/openrouter/models/black-forest-labs/flux.2-pro.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "FLUX.2 Pro" -family = "flux" -release_date = "2025-11-25" -last_updated = "2026-01-31" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = false - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 46_864 -output = 46_864 - -[modalities] -input = ["image","text"] -output = ["image"] diff --git a/providers/openrouter/models/bytedance-seed/seed-1.6-flash.toml b/providers/openrouter/models/bytedance-seed/seed-1.6-flash.toml new file mode 100644 index 000000000..323674f62 --- /dev/null +++ b/providers/openrouter/models/bytedance-seed/seed-1.6-flash.toml @@ -0,0 +1,22 @@ +name = "Seed 1.6 Flash" +family = "seed" +release_date = "2025-12-23" +last_updated = "2025-12-23" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.075 +output = 0.3 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["image", "text", "video"] +output = ["text"] diff --git a/providers/openrouter/models/bytedance-seed/seed-1.6.toml b/providers/openrouter/models/bytedance-seed/seed-1.6.toml new file mode 100644 index 000000000..2e2b626ee --- /dev/null +++ b/providers/openrouter/models/bytedance-seed/seed-1.6.toml @@ -0,0 +1,22 @@ +name = "Seed 1.6" +family = "seed" +release_date = "2025-12-23" +last_updated = "2025-12-23" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.25 +output = 2 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["image", "text", "video"] +output = ["text"] diff --git a/providers/openrouter/models/bytedance-seed/seed-2.0-lite.toml b/providers/openrouter/models/bytedance-seed/seed-2.0-lite.toml new file mode 100644 index 000000000..95c1b3823 --- /dev/null +++ b/providers/openrouter/models/bytedance-seed/seed-2.0-lite.toml @@ -0,0 +1,22 @@ +name = "Seed-2.0-Lite" +family = "seed" +release_date = "2026-03-10" +last_updated = "2026-03-10" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.25 +output = 2 + +[limit] +context = 262_144 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/bytedance-seed/seed-2.0-mini.toml b/providers/openrouter/models/bytedance-seed/seed-2.0-mini.toml new file mode 100644 index 000000000..6679d4912 --- /dev/null +++ b/providers/openrouter/models/bytedance-seed/seed-2.0-mini.toml @@ -0,0 +1,22 @@ +name = "Seed-2.0-Mini" +family = "seed" +release_date = "2026-02-26" +last_updated = "2026-02-26" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.1 +output = 0.4 + +[limit] +context = 262_144 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/bytedance-seed/seedream-4.5.toml b/providers/openrouter/models/bytedance-seed/seedream-4.5.toml deleted file mode 100644 index 69bb02cb5..000000000 --- a/providers/openrouter/models/bytedance-seed/seedream-4.5.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Seedream 4.5" -family = "seed" -release_date = "2025-12-23" -last_updated = "2026-01-31" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 4_096 -output = 4_096 - -[modalities] -input = ["image","text"] -output = ["image"] diff --git a/providers/openrouter/models/bytedance/ui-tars-1.5-7b.toml b/providers/openrouter/models/bytedance/ui-tars-1.5-7b.toml new file mode 100644 index 000000000..1a34b0216 --- /dev/null +++ b/providers/openrouter/models/bytedance/ui-tars-1.5-7b.toml @@ -0,0 +1,23 @@ +name = "UI-TARS 7B " +release_date = "2025-07-22" +last_updated = "2025-07-22" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-01-31" +open_weights = true + +[cost] +input = 0.1 +output = 0.2 +cache_read = 0.1 + +[limit] +context = 128_000 +output = 2_048 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/cognitivecomputations/dolphin-mistral-24b-venice-edition:free.toml b/providers/openrouter/models/cognitivecomputations/dolphin-mistral-24b-venice-edition:free.toml index eaeb5413a..4d03f103e 100644 --- a/providers/openrouter/models/cognitivecomputations/dolphin-mistral-24b-venice-edition:free.toml +++ b/providers/openrouter/models/cognitivecomputations/dolphin-mistral-24b-venice-edition:free.toml @@ -1,19 +1,18 @@ name = "Uncensored (free)" family = "mistral" release_date = "2025-07-09" -last_updated = "2026-01-31" +last_updated = "2025-07-09" attachment = false reasoning = false temperature = true -# may be inaccurate -knowledge = "2025-06" tool_call = false structured_output = true +knowledge = "2024-04-30" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 32_768 diff --git a/providers/openrouter/models/google/gemma-3-12b-it:free.toml b/providers/openrouter/models/cohere/command-a.toml similarity index 53% rename from providers/openrouter/models/google/gemma-3-12b-it:free.toml rename to providers/openrouter/models/cohere/command-a.toml index 2d43d6e29..021af0c27 100644 --- a/providers/openrouter/models/google/gemma-3-12b-it:free.toml +++ b/providers/openrouter/models/cohere/command-a.toml @@ -1,22 +1,23 @@ -name = "Gemma 3 12B (free)" -family = "gemma" +name = "Command A" +family = "command-a" release_date = "2025-03-13" last_updated = "2025-03-13" -attachment = true +attachment = false reasoning = false temperature = true -knowledge = "2024-10" tool_call = false +structured_output = true +knowledge = "2024-08-31" open_weights = true [cost] -input = 0 -output = 0 +input = 2.5 +output = 10 [limit] -context = 32_768 +context = 256_000 output = 8_192 [modalities] -input = ["text", "image"] +input = ["text"] output = ["text"] diff --git a/providers/openrouter/models/cohere/command-r-08-2024.toml b/providers/openrouter/models/cohere/command-r-08-2024.toml new file mode 100644 index 000000000..65ee0c73d --- /dev/null +++ b/providers/openrouter/models/cohere/command-r-08-2024.toml @@ -0,0 +1,23 @@ +name = "Command R (08-2024)" +family = "command-r" +release_date = "2024-08-30" +last_updated = "2024-08-30" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-03-31" +open_weights = false + +[cost] +input = 0.15 +output = 0.6 + +[limit] +context = 128_000 +output = 4_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/cohere/command-r-plus-08-2024.toml b/providers/openrouter/models/cohere/command-r-plus-08-2024.toml new file mode 100644 index 000000000..934add22c --- /dev/null +++ b/providers/openrouter/models/cohere/command-r-plus-08-2024.toml @@ -0,0 +1,23 @@ +name = "Command R+ (08-2024)" +family = "command-r" +release_date = "2024-08-30" +last_updated = "2024-08-30" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-03-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 128_000 +output = 4_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/cohere/command-r7b-12-2024.toml b/providers/openrouter/models/cohere/command-r7b-12-2024.toml new file mode 100644 index 000000000..00c051f43 --- /dev/null +++ b/providers/openrouter/models/cohere/command-r7b-12-2024.toml @@ -0,0 +1,23 @@ +name = "Command R7B (12-2024)" +family = "command-r" +release_date = "2024-12-14" +last_updated = "2024-12-14" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2024-08-31" +open_weights = false + +[cost] +input = 0.0375 +output = 0.15 + +[limit] +context = 128_000 +output = 4_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepcogito/cogito-v2.1-671b.toml b/providers/openrouter/models/deepcogito/cogito-v2.1-671b.toml new file mode 100644 index 000000000..d917a3c6a --- /dev/null +++ b/providers/openrouter/models/deepcogito/cogito-v2.1-671b.toml @@ -0,0 +1,22 @@ +name = "Cogito v2.1 671B" +family = "cogito" +release_date = "2025-11-13" +last_updated = "2025-11-13" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 1.25 +output = 1.25 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-chat-v3-0324.toml b/providers/openrouter/models/deepseek/deepseek-chat-v3-0324.toml index fd50c50a6..c2ba53505 100644 --- a/providers/openrouter/models/deepseek/deepseek-chat-v3-0324.toml +++ b/providers/openrouter/models/deepseek/deepseek-chat-v3-0324.toml @@ -1,4 +1,3 @@ -id = "deepseek/deepseek-chat-v3-0324:free" name = "DeepSeek V3 0324" family = "deepseek" release_date = "2025-03-24" @@ -6,13 +5,20 @@ last_updated = "2025-03-24" attachment = false reasoning = false temperature = true -knowledge = "2024-10" -tool_call = false +tool_call = true structured_output = true +knowledge = "2024-07-31" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 16384, output = 8192 } + +[cost] +input = 0.2 +output = 0.77 +cache_read = 0.135 + +[limit] +context = 163_840 +output = 16_384 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-chat-v3.1.toml b/providers/openrouter/models/deepseek/deepseek-chat-v3.1.toml index 5f34d4e88..1f531d16b 100644 --- a/providers/openrouter/models/deepseek/deepseek-chat-v3.1.toml +++ b/providers/openrouter/models/deepseek/deepseek-chat-v3.1.toml @@ -1,22 +1,23 @@ -name = "DeepSeek-V3.1" +name = "DeepSeek V3.1" family = "deepseek" release_date = "2025-08-21" last_updated = "2025-08-21" attachment = false reasoning = true temperature = true -knowledge = "2025-07" tool_call = true structured_output = true +knowledge = "2025-03-31" open_weights = true [cost] -input = 0.20 -output = 0.80 +input = 0.21 +output = 0.79 +cache_read = 0.13 [limit] context = 163_840 -output = 163_840 +output = 32_768 [modalities] input = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-chat.toml b/providers/openrouter/models/deepseek/deepseek-chat.toml new file mode 100644 index 000000000..13d849c1c --- /dev/null +++ b/providers/openrouter/models/deepseek/deepseek-chat.toml @@ -0,0 +1,23 @@ +name = "DeepSeek V3" +family = "deepseek" +release_date = "2024-12-26" +last_updated = "2024-12-26" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-07-31" +open_weights = true + +[cost] +input = 0.32 +output = 0.89 + +[limit] +context = 163_840 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-r1-0528.toml b/providers/openrouter/models/deepseek/deepseek-r1-0528.toml new file mode 100644 index 000000000..cb9e693c1 --- /dev/null +++ b/providers/openrouter/models/deepseek/deepseek-r1-0528.toml @@ -0,0 +1,24 @@ +name = "R1 0528" +family = "deepseek" +release_date = "2025-05-28" +last_updated = "2025-05-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.5 +output = 2.15 +cache_read = 0.35 + +[limit] +context = 163_840 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-r1-distill-llama-70b.toml b/providers/openrouter/models/deepseek/deepseek-r1-distill-llama-70b.toml index 51f69e9b6..3511db36b 100644 --- a/providers/openrouter/models/deepseek/deepseek-r1-distill-llama-70b.toml +++ b/providers/openrouter/models/deepseek/deepseek-r1-distill-llama-70b.toml @@ -1,18 +1,23 @@ -id = "deepseek/deepseek-r1-distill-llama-70b:free" -name = "DeepSeek R1 Distill Llama 70B" +name = "R1 Distill Llama 70B" family = "deepseek-thinking" release_date = "2025-01-23" last_updated = "2025-01-23" attachment = false reasoning = true temperature = true -knowledge = "2024-10" tool_call = false structured_output = true +knowledge = "2024-07-31" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 8192, output = 8192 } + +[cost] +input = 0.7 +output = 0.8 + +[limit] +context = 131_072 +output = 16_384 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-r1-distill-qwen-32b.toml b/providers/openrouter/models/deepseek/deepseek-r1-distill-qwen-32b.toml new file mode 100644 index 000000000..7fe0362e5 --- /dev/null +++ b/providers/openrouter/models/deepseek/deepseek-r1-distill-qwen-32b.toml @@ -0,0 +1,23 @@ +name = "R1 Distill Qwen 32B" +family = "deepseek" +release_date = "2025-01-29" +last_updated = "2025-01-29" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = true +knowledge = "2024-07-31" +open_weights = true + +[cost] +input = 0.29 +output = 0.29 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-r1.toml b/providers/openrouter/models/deepseek/deepseek-r1.toml index 3fac3fe10..a90c43592 100644 --- a/providers/openrouter/models/deepseek/deepseek-r1.toml +++ b/providers/openrouter/models/deepseek/deepseek-r1.toml @@ -1,18 +1,18 @@ -name = "DeepSeek: R1" +name = "R1" family = "deepseek-thinking" release_date = "2025-01-20" last_updated = "2025-01-20" attachment = false reasoning = true temperature = true -knowledge = "2024-07" tool_call = true structured_output = false +knowledge = "2024-07-31" open_weights = true [cost] -input = 0.70 -output = 2.50 +input = 0.7 +output = 2.5 [limit] context = 64_000 diff --git a/providers/openrouter/models/deepseek/deepseek-v3.1-terminus.toml b/providers/openrouter/models/deepseek/deepseek-v3.1-terminus.toml index c0cb56e84..75fcf9b34 100644 --- a/providers/openrouter/models/deepseek/deepseek-v3.1-terminus.toml +++ b/providers/openrouter/models/deepseek/deepseek-v3.1-terminus.toml @@ -5,18 +5,19 @@ last_updated = "2025-09-22" attachment = false reasoning = true temperature = true -knowledge = "2025-07" tool_call = true structured_output = true +knowledge = "2025-03-31" open_weights = true [cost] input = 0.27 -output = 1.00 +output = 0.95 +cache_read = 0.13 [limit] -context = 131_072 -output = 65_536 +context = 163_840 +output = 32_768 [modalities] input = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-v3.1-terminus:exacto.toml b/providers/openrouter/models/deepseek/deepseek-v3.2-exp.toml similarity index 60% rename from providers/openrouter/models/deepseek/deepseek-v3.1-terminus:exacto.toml rename to providers/openrouter/models/deepseek/deepseek-v3.2-exp.toml index 5350b584d..9d4885ca7 100644 --- a/providers/openrouter/models/deepseek/deepseek-v3.1-terminus:exacto.toml +++ b/providers/openrouter/models/deepseek/deepseek-v3.2-exp.toml @@ -1,21 +1,21 @@ -name = "DeepSeek V3.1 Terminus (exacto)" +name = "DeepSeek V3.2 Exp" family = "deepseek" -release_date = "2025-09-22" -last_updated = "2025-09-22" +release_date = "2025-09-29" +last_updated = "2025-09-29" attachment = false reasoning = true temperature = true -knowledge = "2025-07" tool_call = true structured_output = true +knowledge = "2025-07-31" open_weights = true [cost] input = 0.27 -output = 1.00 +output = 0.41 [limit] -context = 131_072 +context = 163_840 output = 65_536 [modalities] diff --git a/providers/openrouter/models/deepseek/deepseek-v3.2-speciale.toml b/providers/openrouter/models/deepseek/deepseek-v3.2-speciale.toml index c725c66d5..b2c192ba1 100644 --- a/providers/openrouter/models/deepseek/deepseek-v3.2-speciale.toml +++ b/providers/openrouter/models/deepseek/deepseek-v3.2-speciale.toml @@ -5,18 +5,19 @@ last_updated = "2025-12-01" attachment = false reasoning = true temperature = true -knowledge = "2024-07" -tool_call = true +tool_call = false structured_output = true +knowledge = "2024-07" open_weights = true [cost] -input = 0.27 -output = 0.41 +input = 0.287 +output = 0.431 +cache_read = 0.058 [limit] context = 163_840 -output = 65_536 +output = 163_840 [modalities] input = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-v3.2.toml b/providers/openrouter/models/deepseek/deepseek-v3.2.toml index df05be8cb..afc4dab1a 100644 --- a/providers/openrouter/models/deepseek/deepseek-v3.2.toml +++ b/providers/openrouter/models/deepseek/deepseek-v3.2.toml @@ -5,17 +5,18 @@ last_updated = "2025-12-01" attachment = false reasoning = true temperature = true -knowledge = "2024-07" tool_call = true structured_output = true +knowledge = "2024-07" open_weights = true [cost] -input = 0.28 -output = 0.40 +input = 0.252 +output = 0.378 +cache_read = 0.0252 [limit] -context = 163_840 +context = 131_072 output = 65_536 [modalities] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index c3e5156a3..8cd602449 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -1,11 +1,26 @@ +name = "DeepSeek V4 Flash" +family = "deepseek" +release_date = "2026-04-24" +last_updated = "2026-04-24" attachment = false - -[extends] -from = "deepseek/deepseek-v4-flash" +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true [interleaved] field = "reasoning_content" +[cost] +input = 0.126 +output = 0.252 +cache_read = 0.0252 + [limit] context = 1_048_576 -output = 393_216 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash:free.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash:free.toml new file mode 100644 index 000000000..60b65dbfb --- /dev/null +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash:free.toml @@ -0,0 +1,22 @@ +name = "DeepSeek V4 Flash (free)" +family = "deepseek" +release_date = "2026-04-24" +last_updated = "2026-04-24" +attachment = false +reasoning = true +temperature = false +tool_call = true +structured_output = false +open_weights = true + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 +output = 384_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index ee704136a..866acee4f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -1,11 +1,26 @@ +name = "DeepSeek V4 Pro" +family = "deepseek" +release_date = "2026-04-24" +last_updated = "2026-04-24" attachment = false - -[extends] -from = "deepseek/deepseek-v4-pro" +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true [interleaved] field = "reasoning_content" +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.003625 + [limit] context = 1_048_576 -output = 393_216 +output = 384_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/essentialai/rnj-1-instruct.toml b/providers/openrouter/models/essentialai/rnj-1-instruct.toml new file mode 100644 index 000000000..41025857a --- /dev/null +++ b/providers/openrouter/models/essentialai/rnj-1-instruct.toml @@ -0,0 +1,22 @@ +name = "Rnj 1 Instruct" +family = "rnj" +release_date = "2025-12-07" +last_updated = "2025-12-07" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.15 +output = 0.15 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.0-flash-001.toml b/providers/openrouter/models/google/gemini-2.0-flash-001.toml index 382504c73..7d75fa4fa 100644 --- a/providers/openrouter/models/google/gemini-2.0-flash-001.toml +++ b/providers/openrouter/models/google/gemini-2.0-flash-001.toml @@ -1,24 +1,25 @@ name = "Gemini 2.0 Flash" family = "gemini-flash" -release_date = "2024-12-11" -last_updated = "2024-12-11" +release_date = "2025-02-05" +last_updated = "2025-02-05" attachment = true reasoning = false temperature = true -knowledge = "2024-06" tool_call = true structured_output = true +knowledge = "2024-08-31" open_weights = false [cost] -input = 0.10 -output = 0.40 +input = 0.1 +output = 0.4 cache_read = 0.025 +cache_write = 0.083333 [limit] context = 1_048_576 output = 8_192 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "pdf", "audio", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.0-flash-lite-001.toml b/providers/openrouter/models/google/gemini-2.0-flash-lite-001.toml new file mode 100644 index 000000000..2a12e9cae --- /dev/null +++ b/providers/openrouter/models/google/gemini-2.0-flash-lite-001.toml @@ -0,0 +1,23 @@ +name = "Gemini 2.0 Flash Lite" +family = "gemini" +release_date = "2025-02-25" +last_updated = "2025-02-25" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-08-31" +open_weights = false + +[cost] +input = 0.075 +output = 0.3 + +[limit] +context = 1_048_576 +output = 8_192 + +[modalities] +input = ["text", "image", "pdf", "audio", "video"] +output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.5-flash-image.toml b/providers/openrouter/models/google/gemini-2.5-flash-image.toml new file mode 100644 index 000000000..ce0ea749c --- /dev/null +++ b/providers/openrouter/models/google/gemini-2.5-flash-image.toml @@ -0,0 +1,25 @@ +name = "Nano Banana (Gemini 2.5 Flash Image)" +family = "gemini" +release_date = "2025-10-07" +last_updated = "2025-10-07" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2025-01-31" +open_weights = false + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 +cache_write = 0.083333 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["image", "text"] +output = ["image", "text"] diff --git a/providers/openrouter/models/google/gemini-2.5-flash-lite-preview-09-2025.toml b/providers/openrouter/models/google/gemini-2.5-flash-lite-preview-09-2025.toml index 92277d387..367e1a20d 100644 --- a/providers/openrouter/models/google/gemini-2.5-flash-lite-preview-09-2025.toml +++ b/providers/openrouter/models/google/gemini-2.5-flash-lite-preview-09-2025.toml @@ -1,24 +1,26 @@ -name = "Gemini 2.5 Flash Lite Preview 09-25" +name = "Gemini 2.5 Flash Lite Preview 09-2025" family = "gemini-flash-lite" release_date = "2025-09-25" last_updated = "2025-09-25" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01-31" open_weights = false [cost] -input = 0.10 -output = 0.40 -cache_read = 0.025 +input = 0.1 +output = 0.4 +reasoning = 0.4 +cache_read = 0.01 +cache_write = 0.083333 [limit] context = 1_048_576 -output = 65_536 +output = 65_535 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "pdf", "audio", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.5-flash-lite.toml b/providers/openrouter/models/google/gemini-2.5-flash-lite.toml index b310575dd..4a68371b0 100644 --- a/providers/openrouter/models/google/gemini-2.5-flash-lite.toml +++ b/providers/openrouter/models/google/gemini-2.5-flash-lite.toml @@ -1,24 +1,26 @@ name = "Gemini 2.5 Flash Lite" family = "gemini-flash-lite" -release_date = "2025-06-17" -last_updated = "2025-06-17" +release_date = "2025-07-22" +last_updated = "2025-07-22" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01-31" open_weights = false [cost] -input = 0.10 -output = 0.40 -cache_read = 0.025 +input = 0.1 +output = 0.4 +reasoning = 0.4 +cache_read = 0.01 +cache_write = 0.083333 [limit] context = 1_048_576 -output = 65_536 +output = 65_535 [modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] \ No newline at end of file +input = ["text", "image", "pdf", "audio", "video"] +output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.5-flash.toml b/providers/openrouter/models/google/gemini-2.5-flash.toml index 0aa2151e7..2fea51145 100644 --- a/providers/openrouter/models/google/gemini-2.5-flash.toml +++ b/providers/openrouter/models/google/gemini-2.5-flash.toml @@ -1,24 +1,26 @@ name = "Gemini 2.5 Flash" family = "gemini-flash" -release_date = "2025-07-17" -last_updated = "2025-07-17" +release_date = "2025-06-17" +last_updated = "2025-06-17" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01-31" open_weights = false [cost] -input = 0.30 -output = 2.50 -cache_read = 0.0375 +input = 0.3 +output = 2.5 +reasoning = 2.5 +cache_read = 0.03 +cache_write = 0.083333 [limit] context = 1_048_576 -output = 65_536 +output = 65_535 [modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] \ No newline at end of file +input = ["pdf", "image", "text", "audio", "video"] +output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.5-pro-preview-05-06.toml b/providers/openrouter/models/google/gemini-2.5-pro-preview-05-06.toml index 4645029df..777e26777 100644 --- a/providers/openrouter/models/google/gemini-2.5-pro-preview-05-06.toml +++ b/providers/openrouter/models/google/gemini-2.5-pro-preview-05-06.toml @@ -1,24 +1,26 @@ name = "Gemini 2.5 Pro Preview 05-06" family = "gemini-pro" -release_date = "2025-05-06" -last_updated = "2025-05-06" +release_date = "2025-05-07" +last_updated = "2025-05-07" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01-31" open_weights = false [cost] input = 1.25 -output = 10.00 -cache_read = 0.31 +output = 10 +reasoning = 10 +cache_read = 0.125 +cache_write = 0.375 [limit] context = 1_048_576 -output = 65_536 +output = 65_535 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "pdf", "audio", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.5-pro-preview-06-05.toml b/providers/openrouter/models/google/gemini-2.5-pro-preview.toml similarity index 67% rename from providers/openrouter/models/google/gemini-2.5-pro-preview-06-05.toml rename to providers/openrouter/models/google/gemini-2.5-pro-preview.toml index 156c4d936..4a13a7ab7 100644 --- a/providers/openrouter/models/google/gemini-2.5-pro-preview-06-05.toml +++ b/providers/openrouter/models/google/gemini-2.5-pro-preview.toml @@ -1,24 +1,26 @@ name = "Gemini 2.5 Pro Preview 06-05" -family = "gemini-pro" +family = "gemini" release_date = "2025-06-05" last_updated = "2025-06-05" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01-31" open_weights = false [cost] input = 1.25 -output = 10.00 -cache_read = 0.31 +output = 10 +reasoning = 10 +cache_read = 0.125 +cache_write = 0.375 [limit] context = 1_048_576 output = 65_536 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["pdf", "image", "text", "audio"] output = ["text"] diff --git a/providers/openrouter/models/google/gemini-2.5-pro.toml b/providers/openrouter/models/google/gemini-2.5-pro.toml index bd908b3a3..10202473f 100644 --- a/providers/openrouter/models/google/gemini-2.5-pro.toml +++ b/providers/openrouter/models/google/gemini-2.5-pro.toml @@ -1,2 +1,26 @@ -[extends] -from = "google/gemini-2.5-pro" +name = "Gemini 2.5 Pro" +family = "gemini" +release_date = "2025-06-17" +last_updated = "2025-06-17" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01-31" +open_weights = false + +[cost] +input = 1.25 +output = 10 +reasoning = 10 +cache_read = 0.125 +cache_write = 0.375 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "pdf", "audio", "video"] +output = ["text"] diff --git a/providers/openrouter/models/google/gemini-3-flash-preview.toml b/providers/openrouter/models/google/gemini-3-flash-preview.toml index fa6d1e245..17b0a973f 100644 --- a/providers/openrouter/models/google/gemini-3-flash-preview.toml +++ b/providers/openrouter/models/google/gemini-3-flash-preview.toml @@ -5,23 +5,25 @@ last_updated = "2025-12-17" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = false [interleaved] field = "reasoning_details" [cost] -input = 0.50 -output = 3.00 +input = 0.5 +output = 3 +reasoning = 3 cache_read = 0.05 +cache_write = 0.083333 [limit] context = 1_048_576 output = 65_536 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "pdf", "audio", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemini-3-pro-image-preview.toml b/providers/openrouter/models/google/gemini-3-pro-image-preview.toml new file mode 100644 index 000000000..e88cbbda1 --- /dev/null +++ b/providers/openrouter/models/google/gemini-3-pro-image-preview.toml @@ -0,0 +1,25 @@ +name = "Nano Banana Pro (Gemini 3 Pro Image Preview)" +family = "gemini" +release_date = "2025-11-20" +last_updated = "2025-11-20" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 2 +output = 12 +reasoning = 12 +cache_read = 0.2 +cache_write = 0.375 + +[limit] +context = 65_536 +output = 32_768 + +[modalities] +input = ["image", "text"] +output = ["image", "text"] diff --git a/providers/openrouter/models/google/gemini-3-pro-preview.toml b/providers/openrouter/models/google/gemini-3-pro-preview.toml deleted file mode 100644 index 81523a7d4..000000000 --- a/providers/openrouter/models/google/gemini-3-pro-preview.toml +++ /dev/null @@ -1,26 +0,0 @@ -name = "Gemini 3 Pro Preview" -family = "gemini-pro" -release_date = "2025-11-18" -last_updated = "2025-11" -attachment = true -reasoning = true -temperature = true -knowledge = "2025-01" -tool_call = true -structured_output = true -open_weights = false - -[interleaved] -field = "reasoning_details" - -[cost] -input = 2.00 -output = 12.00 - -[limit] -context = 1_050_000 -output = 66_000 - -[modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] diff --git a/providers/openrouter/models/google/gemini-3.1-flash-image-preview.toml b/providers/openrouter/models/google/gemini-3.1-flash-image-preview.toml index 056a86bbc..cbb2c524a 100644 --- a/providers/openrouter/models/google/gemini-3.1-flash-image-preview.toml +++ b/providers/openrouter/models/google/gemini-3.1-flash-image-preview.toml @@ -1,23 +1,23 @@ -name = "Gemini 3.1 Flash Image Preview (Nano Banana 2)" +name = "Nano Banana 2 (Gemini 3.1 Flash Image Preview)" family = "gemini-flash" release_date = "2026-02-26" last_updated = "2026-02-26" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = false structured_output = true +knowledge = "2025-01" open_weights = false [cost] -input = 0.50 -output = 3.00 +input = 0.5 +output = 3 [limit] context = 65_536 output = 65_536 [modalities] -input = ["text", "image"] -output = ["text", "image"] +input = ["image", "text"] +output = ["image", "text"] diff --git a/providers/openrouter/models/google/gemini-3.1-flash-lite-preview.toml b/providers/openrouter/models/google/gemini-3.1-flash-lite-preview.toml index 3e46b4c6c..dd899fb80 100644 --- a/providers/openrouter/models/google/gemini-3.1-flash-lite-preview.toml +++ b/providers/openrouter/models/google/gemini-3.1-flash-lite-preview.toml @@ -14,13 +14,11 @@ input = 0.25 output = 1.5 reasoning = 1.5 cache_read = 0.025 -cache_write = 0.083 -input_audio = 0.5 -output_audio = 0.5 +cache_write = 0.083333 [limit] -context = 1048576 -output = 65536 +context = 1_048_576 +output = 65_536 [modalities] input = ["text", "image", "video", "pdf", "audio"] diff --git a/providers/openrouter/models/google/gemini-3.1-flash-lite.toml b/providers/openrouter/models/google/gemini-3.1-flash-lite.toml new file mode 100644 index 000000000..2b9d1508e --- /dev/null +++ b/providers/openrouter/models/google/gemini-3.1-flash-lite.toml @@ -0,0 +1,25 @@ +name = "Gemini 3.1 Flash Lite" +family = "gemini" +release_date = "2026-05-07" +last_updated = "2026-05-07" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.25 +output = 1.5 +reasoning = 1.5 +cache_read = 0.025 +cache_write = 0.083333 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "video", "pdf", "audio"] +output = ["text"] diff --git a/providers/openrouter/models/google/gemini-3.1-pro-preview-customtools.toml b/providers/openrouter/models/google/gemini-3.1-pro-preview-customtools.toml index 3be0876e0..dc9405304 100644 --- a/providers/openrouter/models/google/gemini-3.1-pro-preview-customtools.toml +++ b/providers/openrouter/models/google/gemini-3.1-pro-preview-customtools.toml @@ -1,33 +1,35 @@ name = "Gemini 3.1 Pro Preview Custom Tools" family = "gemini-pro" -release_date = "2026-02-19" -last_updated = "2026-02-19" +release_date = "2026-02-25" +last_updated = "2026-02-25" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = false [interleaved] field = "reasoning_details" [cost] -input = 2.00 -output = 12.00 -reasoning = 12.00 +input = 2 +output = 12 +reasoning = 12 +cache_read = 0.2 +cache_write = 0.375 [[cost.tiers]] tier = { size = 200_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 +input = 4 +output = 18 +cache_read = 0.4 [limit] context = 1_048_576 output = 65_536 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "audio", "image", "video", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/google/gemini-3.1-pro-preview.toml b/providers/openrouter/models/google/gemini-3.1-pro-preview.toml index 0dd2e7f6d..0e030f93b 100644 --- a/providers/openrouter/models/google/gemini-3.1-pro-preview.toml +++ b/providers/openrouter/models/google/gemini-3.1-pro-preview.toml @@ -5,29 +5,31 @@ last_updated = "2026-02-19" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = false [interleaved] field = "reasoning_details" [cost] -input = 2.00 -output = 12.00 -reasoning = 12.00 +input = 2 +output = 12 +reasoning = 12 +cache_read = 0.2 +cache_write = 0.375 [[cost.tiers]] tier = { size = 200_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 +input = 4 +output = 18 +cache_read = 0.4 [limit] context = 1_048_576 output = 65_536 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["audio", "pdf", "image", "text", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemma-2-9b-it.toml b/providers/openrouter/models/google/gemma-2-27b-it.toml similarity index 53% rename from providers/openrouter/models/google/gemma-2-9b-it.toml rename to providers/openrouter/models/google/gemma-2-27b-it.toml index 947f07f70..c59ff9fd1 100644 --- a/providers/openrouter/models/google/gemma-2-9b-it.toml +++ b/providers/openrouter/models/google/gemma-2-27b-it.toml @@ -1,21 +1,22 @@ -name = "Gemma 2 9B" +name = "Gemma 2 27B" family = "gemma" -release_date = "2024-06-28" -last_updated = "2024-06-28" +release_date = "2024-07-13" +last_updated = "2024-07-13" attachment = false reasoning = false temperature = true -knowledge = "2024-06" tool_call = false +structured_output = true +knowledge = "2024-06-30" open_weights = true [cost] -input = 0.03 -output = 0.09 +input = 0.65 +output = 0.65 [limit] context = 8_192 -output = 8_192 +output = 2_048 [modalities] input = ["text"] diff --git a/providers/openrouter/models/google/gemma-3-12b-it.toml b/providers/openrouter/models/google/gemma-3-12b-it.toml index 73a26d10f..1065b1b1a 100644 --- a/providers/openrouter/models/google/gemma-3-12b-it.toml +++ b/providers/openrouter/models/google/gemma-3-12b-it.toml @@ -5,18 +5,18 @@ last_updated = "2025-03-13" attachment = true reasoning = false temperature = true -knowledge = "2024-10" -tool_call = false +tool_call = true structured_output = true +knowledge = "2024-08-31" open_weights = true [cost] -input = 0.03 -output = 0.10 +input = 0.04 +output = 0.13 [limit] context = 131_072 -output = 131_072 +output = 16_384 [modalities] input = ["text", "image"] diff --git a/providers/openrouter/models/google/gemma-3-27b-it.toml b/providers/openrouter/models/google/gemma-3-27b-it.toml index d7ee6272d..07e8a68be 100644 --- a/providers/openrouter/models/google/gemma-3-27b-it.toml +++ b/providers/openrouter/models/google/gemma-3-27b-it.toml @@ -5,18 +5,18 @@ last_updated = "2025-03-12" attachment = true reasoning = false temperature = true -knowledge = "2024-10" tool_call = true structured_output = true +knowledge = "2024-08-31" open_weights = true [cost] -input = 0.04 -output = 0.15 +input = 0.08 +output = 0.16 [limit] -context = 96_000 -output = 96_000 +context = 131_072 +output = 16_384 [modalities] input = ["text", "image"] diff --git a/providers/openrouter/models/google/gemma-3-27b-it:free.toml b/providers/openrouter/models/google/gemma-3-27b-it:free.toml deleted file mode 100644 index 2c581a922..000000000 --- a/providers/openrouter/models/google/gemma-3-27b-it:free.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Gemma 3 27B (free)" -family = "gemma" -release_date = "2025-03-12" -last_updated = "2025-03-12" -attachment = true -reasoning = false -temperature = true -knowledge = "2024-10" -tool_call = true -open_weights = true - -[cost] -input = 0 -output = 0 - -[limit] -context = 131_072 -output = 8_192 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/openrouter/models/google/gemma-3-4b-it.toml b/providers/openrouter/models/google/gemma-3-4b-it.toml index e678c830e..97a65f7a7 100644 --- a/providers/openrouter/models/google/gemma-3-4b-it.toml +++ b/providers/openrouter/models/google/gemma-3-4b-it.toml @@ -5,17 +5,18 @@ last_updated = "2025-03-13" attachment = true reasoning = false temperature = true -knowledge = "2024-10" tool_call = false +structured_output = true +knowledge = "2024-08-31" open_weights = true [cost] -input = 0.01703 -output = 0.06815 +input = 0.04 +output = 0.08 [limit] -context = 96_000 -output = 96_000 +context = 131_072 +output = 16_384 [modalities] input = ["text", "image"] diff --git a/providers/openrouter/models/google/gemma-3-4b-it:free.toml b/providers/openrouter/models/google/gemma-3-4b-it:free.toml deleted file mode 100644 index 9a10b0f97..000000000 --- a/providers/openrouter/models/google/gemma-3-4b-it:free.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Gemma 3 4B (free)" -family = "gemma" -release_date = "2025-03-13" -last_updated = "2025-03-13" -attachment = true -reasoning = false -temperature = true -knowledge = "2024-10" -tool_call = false -open_weights = true - -[cost] -input = 0 -output = 0 - -[limit] -context = 32_768 -output = 8_192 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/openrouter/models/google/gemma-3n-e2b-it:free.toml b/providers/openrouter/models/google/gemma-3n-e2b-it:free.toml deleted file mode 100644 index e876ed841..000000000 --- a/providers/openrouter/models/google/gemma-3n-e2b-it:free.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Gemma 3n 2B (free)" -family = "gemma" -release_date = "2025-07-09" -last_updated = "2025-07-09" -attachment = true -reasoning = false -temperature = true -knowledge = "2024-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 8_192 -output = 2_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/openrouter/models/google/gemma-3n-e4b-it.toml b/providers/openrouter/models/google/gemma-3n-e4b-it.toml index 5418c69b6..0cff34677 100644 --- a/providers/openrouter/models/google/gemma-3n-e4b-it.toml +++ b/providers/openrouter/models/google/gemma-3n-e4b-it.toml @@ -2,16 +2,17 @@ name = "Gemma 3n 4B" family = "gemma" release_date = "2025-05-20" last_updated = "2025-05-20" -attachment = true +attachment = false reasoning = false temperature = true -knowledge = "2024-06" tool_call = false +structured_output = false +knowledge = "2024-08-31" open_weights = true -[cost] -input = 0.02 -output = 0.04 +[cost] +input = 0.06 +output = 0.12 [limit] context = 32_768 diff --git a/providers/openrouter/models/google/gemma-3n-e4b-it:free.toml b/providers/openrouter/models/google/gemma-3n-e4b-it:free.toml deleted file mode 100644 index f057e41ed..000000000 --- a/providers/openrouter/models/google/gemma-3n-e4b-it:free.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Gemma 3n 4B (free)" -family = "gemma" -release_date = "2025-05-20" -last_updated = "2025-05-20" -attachment = true -reasoning = false -temperature = true -knowledge = "2024-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 8_192 -output = 2_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml b/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml index cae875442..518e7f0cc 100644 --- a/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml +++ b/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml @@ -1,24 +1,23 @@ -name = "Gemma 4 26B A4B" +name = "Gemma 4 26B A4B " family = "gemma" release_date = "2026-04-03" last_updated = "2026-04-03" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = true [cost] -input = 0.13 -output = 0.40 +input = 0.06 +output = 0.33 [limit] context = 262_144 output = 262_144 [modalities] -input = ["text", "image", "video"] +input = ["image", "text", "video"] output = ["text"] - diff --git a/providers/openrouter/models/google/gemma-4-26b-a4b-it:free.toml b/providers/openrouter/models/google/gemma-4-26b-a4b-it:free.toml index 6229f0b48..d7a0cf608 100644 --- a/providers/openrouter/models/google/gemma-4-26b-a4b-it:free.toml +++ b/providers/openrouter/models/google/gemma-4-26b-a4b-it:free.toml @@ -1,13 +1,13 @@ -name = "Gemma 4 26B A4B (free)" +name = "Gemma 4 26B A4B (free)" family = "gemma" release_date = "2026-04-03" last_updated = "2026-04-03" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = true [cost] @@ -19,5 +19,5 @@ context = 262_144 output = 32_768 [modalities] -input = ["text", "image", "video"] +input = ["image", "text", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemma-4-31b-it.toml b/providers/openrouter/models/google/gemma-4-31b-it.toml index 9bdc2d3d1..e2ac7e5a2 100644 --- a/providers/openrouter/models/google/gemma-4-31b-it.toml +++ b/providers/openrouter/models/google/gemma-4-31b-it.toml @@ -5,19 +5,19 @@ last_updated = "2026-04-02" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = true [cost] -input = 0.14 -output = 0.40 +input = 0.12 +output = 0.37 [limit] context = 262_144 -output = 262_144 +output = 16_384 [modalities] -input = ["text", "image", "video"] +input = ["image", "text", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/gemma-4-31b-it:free.toml b/providers/openrouter/models/google/gemma-4-31b-it:free.toml index bd7def45c..01c24f0d9 100644 --- a/providers/openrouter/models/google/gemma-4-31b-it:free.toml +++ b/providers/openrouter/models/google/gemma-4-31b-it:free.toml @@ -5,9 +5,9 @@ last_updated = "2026-04-02" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true +knowledge = "2025-01" open_weights = true [cost] @@ -19,5 +19,5 @@ context = 262_144 output = 32_768 [modalities] -input = ["text", "image", "video"] +input = ["image", "text", "video"] output = ["text"] diff --git a/providers/openrouter/models/google/lyria-3-clip-preview.toml b/providers/openrouter/models/google/lyria-3-clip-preview.toml new file mode 100644 index 000000000..006255fed --- /dev/null +++ b/providers/openrouter/models/google/lyria-3-clip-preview.toml @@ -0,0 +1,22 @@ +name = "Lyria 3 Clip Preview" +family = "lyria" +release_date = "2026-03-30" +last_updated = "2026-03-30" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text", "audio"] diff --git a/providers/openrouter/models/google/lyria-3-pro-preview.toml b/providers/openrouter/models/google/lyria-3-pro-preview.toml new file mode 100644 index 000000000..4f9dafbcb --- /dev/null +++ b/providers/openrouter/models/google/lyria-3-pro-preview.toml @@ -0,0 +1,22 @@ +name = "Lyria 3 Pro Preview" +family = "lyria" +release_date = "2026-03-30" +last_updated = "2026-03-30" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text", "audio"] diff --git a/providers/openrouter/models/gryphe/mythomax-l2-13b.toml b/providers/openrouter/models/gryphe/mythomax-l2-13b.toml new file mode 100644 index 000000000..e6adf753d --- /dev/null +++ b/providers/openrouter/models/gryphe/mythomax-l2-13b.toml @@ -0,0 +1,23 @@ +name = "MythoMax 13B" +family = "o" +release_date = "2023-07-02" +last_updated = "2023-07-02" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-06-30" +open_weights = true + +[cost] +input = 0.06 +output = 0.06 + +[limit] +context = 4_096 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/ibm-granite/granite-4.0-h-micro.toml b/providers/openrouter/models/ibm-granite/granite-4.0-h-micro.toml new file mode 100644 index 000000000..f71a0ba0b --- /dev/null +++ b/providers/openrouter/models/ibm-granite/granite-4.0-h-micro.toml @@ -0,0 +1,22 @@ +name = "Granite 4.0 Micro" +family = "granite" +release_date = "2025-10-20" +last_updated = "2025-10-20" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = true + +[cost] +input = 0.017 +output = 0.11 + +[limit] +context = 131_000 +output = 131_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/ibm-granite/granite-4.1-8b.toml b/providers/openrouter/models/ibm-granite/granite-4.1-8b.toml new file mode 100644 index 000000000..5e6bb125c --- /dev/null +++ b/providers/openrouter/models/ibm-granite/granite-4.1-8b.toml @@ -0,0 +1,23 @@ +name = "Granite 4.1 8B" +family = "granite" +release_date = "2026-04-30" +last_updated = "2026-04-30" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.05 +output = 0.1 +cache_read = 0.05 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/inception/mercury-edit-2.toml b/providers/openrouter/models/inception/mercury-edit-2.toml deleted file mode 100644 index d40bbcdf6..000000000 --- a/providers/openrouter/models/inception/mercury-edit-2.toml +++ /dev/null @@ -1,21 +0,0 @@ -name = "Mercury Edit 2" -release_date = "2026-03-30" -last_updated = "2026-03-30" -attachment = false -reasoning = true -temperature = true -tool_call = false -open_weights = false - -[cost] -input = 0.25 -output = 0.75 -cache_read = 0.025 - -[limit] -context = 128000 -output = 8192 - -[modalities] -input = ["text"] -output = ["text"] \ No newline at end of file diff --git a/providers/openrouter/models/inclusionai/ling-2.6-1t.toml b/providers/openrouter/models/inclusionai/ling-2.6-1t.toml new file mode 100644 index 000000000..02688ea2f --- /dev/null +++ b/providers/openrouter/models/inclusionai/ling-2.6-1t.toml @@ -0,0 +1,23 @@ +name = "Ling-2.6-1T" +family = "ling" +release_date = "2026-04-23" +last_updated = "2026-04-23" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.06 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/inclusionai/ling-2.6-flash.toml b/providers/openrouter/models/inclusionai/ling-2.6-flash.toml new file mode 100644 index 000000000..01fa3ae11 --- /dev/null +++ b/providers/openrouter/models/inclusionai/ling-2.6-flash.toml @@ -0,0 +1,23 @@ +name = "Ling-2.6-flash" +family = "ling" +release_date = "2026-04-21" +last_updated = "2026-04-21" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.01 +output = 0.03 +cache_read = 0.002 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/inclusionai/ring-2.6-1t:free.toml b/providers/openrouter/models/inclusionai/ring-2.6-1t:free.toml new file mode 100644 index 000000000..0b13e0ab5 --- /dev/null +++ b/providers/openrouter/models/inclusionai/ring-2.6-1t:free.toml @@ -0,0 +1,22 @@ +name = "Ring-2.6-1T (free)" +family = "ring" +release_date = "2026-05-08" +last_updated = "2026-05-08" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/inflection/inflection-3-pi.toml b/providers/openrouter/models/inflection/inflection-3-pi.toml new file mode 100644 index 000000000..f3e284ee9 --- /dev/null +++ b/providers/openrouter/models/inflection/inflection-3-pi.toml @@ -0,0 +1,23 @@ +name = "Inflection 3 Pi" +family = "o" +release_date = "2024-10-11" +last_updated = "2024-10-11" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2024-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 8_000 +output = 1_024 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/inflection/inflection-3-productivity.toml b/providers/openrouter/models/inflection/inflection-3-productivity.toml new file mode 100644 index 000000000..c04df3885 --- /dev/null +++ b/providers/openrouter/models/inflection/inflection-3-productivity.toml @@ -0,0 +1,23 @@ +name = "Inflection 3 Productivity" +family = "o" +release_date = "2024-10-11" +last_updated = "2024-10-11" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2024-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 8_000 +output = 1_024 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml b/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml new file mode 100644 index 000000000..e5145ea19 --- /dev/null +++ b/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml @@ -0,0 +1,23 @@ +name = "KAT-Coder-Pro V2" +family = "kat-coder" +release_date = "2026-03-27" +last_updated = "2026-03-27" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.06 + +[limit] +context = 256_000 +output = 80_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/liquid/lfm-2-24b-a2b.toml b/providers/openrouter/models/liquid/lfm-2-24b-a2b.toml new file mode 100644 index 000000000..b16fa94a0 --- /dev/null +++ b/providers/openrouter/models/liquid/lfm-2-24b-a2b.toml @@ -0,0 +1,22 @@ +name = "LFM2-24B-A2B" +family = "liquid" +release_date = "2026-02-25" +last_updated = "2026-02-25" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = true + +[cost] +input = 0.03 +output = 0.12 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/liquid/lfm-2.5-1.2b-instruct:free.toml b/providers/openrouter/models/liquid/lfm-2.5-1.2b-instruct:free.toml index abb2facb4..d47a636b3 100644 --- a/providers/openrouter/models/liquid/lfm-2.5-1.2b-instruct:free.toml +++ b/providers/openrouter/models/liquid/lfm-2.5-1.2b-instruct:free.toml @@ -1,21 +1,21 @@ name = "LFM2.5-1.2B-Instruct (free)" family = "liquid" release_date = "2026-01-20" -last_updated = "2026-01-28" +last_updated = "2026-01-20" attachment = false reasoning = false temperature = true -# may be inaccurate -knowledge = "2025-06" tool_call = false +structured_output = false +knowledge = "2025-06" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] -context = 131_072 +context = 32_768 output = 32_768 [modalities] diff --git a/providers/openrouter/models/liquid/lfm-2.5-1.2b-thinking:free.toml b/providers/openrouter/models/liquid/lfm-2.5-1.2b-thinking:free.toml index 220cac22e..7fb5d4900 100644 --- a/providers/openrouter/models/liquid/lfm-2.5-1.2b-thinking:free.toml +++ b/providers/openrouter/models/liquid/lfm-2.5-1.2b-thinking:free.toml @@ -1,21 +1,21 @@ name = "LFM2.5-1.2B-Thinking (free)" family = "liquid" release_date = "2026-01-20" -last_updated = "2026-01-28" +last_updated = "2026-01-20" attachment = false reasoning = true temperature = true -# may be inaccurate -knowledge = "2025-06" tool_call = false +structured_output = false +knowledge = "2025-06" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] -context = 131_072 +context = 32_768 output = 32_768 [modalities] diff --git a/providers/openrouter/models/mancer/weaver.toml b/providers/openrouter/models/mancer/weaver.toml new file mode 100644 index 000000000..0e5555e63 --- /dev/null +++ b/providers/openrouter/models/mancer/weaver.toml @@ -0,0 +1,23 @@ +name = "Weaver (alpha)" +family = "alpha" +release_date = "2023-08-02" +last_updated = "2023-08-02" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-06-30" +open_weights = false + +[cost] +input = 0.75 +output = 1 + +[limit] +context = 8_000 +output = 2_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3-70b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3-70b-instruct.toml new file mode 100644 index 000000000..61b80dbce --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3-70b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3 70B Instruct" +family = "llama" +release_date = "2024-04-18" +last_updated = "2024-04-18" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.51 +output = 0.74 + +[limit] +context = 8_192 +output = 8_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3-8b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3-8b-instruct.toml new file mode 100644 index 000000000..e302e4379 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3-8b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3 8B Instruct" +family = "llama" +release_date = "2024-04-18" +last_updated = "2024-04-18" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.04 +output = 0.04 + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml new file mode 100644 index 000000000..22218c329 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3.1 70B Instruct" +family = "llama" +release_date = "2024-07-23" +last_updated = "2024-07-23" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.4 +output = 0.4 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.1-8b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.1-8b-instruct.toml new file mode 100644 index 000000000..bb78b2500 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3.1-8b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3.1 8B Instruct" +family = "llama" +release_date = "2024-07-23" +last_updated = "2024-07-23" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.02 +output = 0.05 + +[limit] +context = 16_384 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.2-11b-vision-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.2-11b-vision-instruct.toml index 123e73b77..6e99c7ed3 100644 --- a/providers/openrouter/models/meta-llama/llama-3.2-11b-vision-instruct.toml +++ b/providers/openrouter/models/meta-llama/llama-3.2-11b-vision-instruct.toml @@ -1,4 +1,3 @@ -id = "meta-llama/llama-3.2-11b-vision-instruct:free" name = "Llama 3.2 11B Vision Instruct" family = "llama" release_date = "2024-09-25" @@ -6,12 +5,19 @@ last_updated = "2024-09-25" attachment = true reasoning = false temperature = true -knowledge = "2023-12" tool_call = false +structured_output = true +knowledge = "2023-12-31" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 131072, output = 8192 } + +[cost] +input = 0.245 +output = 0.245 + +[limit] +context = 131_072 +output = 16_384 [modalities] input = ["text", "image"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.2-1b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.2-1b-instruct.toml new file mode 100644 index 000000000..4407883fe --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3.2-1b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3.2 1B Instruct" +family = "llama" +release_date = "2024-09-25" +last_updated = "2024-09-25" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.027 +output = 0.2 + +[limit] +context = 60_000 +output = 60_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct.toml new file mode 100644 index 000000000..19401cdd0 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3.2 3B Instruct" +family = "llama" +release_date = "2024-09-25" +last_updated = "2024-09-25" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.051 +output = 0.34 + +[limit] +context = 80_000 +output = 80_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct:free.toml b/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct:free.toml index 40211ace3..227f4d609 100644 --- a/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct:free.toml +++ b/providers/openrouter/models/meta-llama/llama-3.2-3b-instruct:free.toml @@ -2,15 +2,22 @@ name = "Llama 3.2 3B Instruct (free)" family = "llama" release_date = "2024-09-25" last_updated = "2024-09-25" -attachment = true +attachment = false reasoning = false temperature = true -knowledge = "2023-12" tool_call = false +structured_output = false +knowledge = "2023-12-31" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 131072, output = 131072 } + +[cost] +input = 0 +output = 0 + +[limit] +context = 131_072 +output = 131_072 [modalities] -input = ["text", "image"] -output = ["text"] \ No newline at end of file +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct.toml new file mode 100644 index 000000000..174316e38 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct.toml @@ -0,0 +1,23 @@ +name = "Llama 3.3 70B Instruct" +family = "llama" +release_date = "2024-12-06" +last_updated = "2024-12-06" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.1 +output = 0.32 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct:free.toml b/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct:free.toml index 006f17223..d6570c6f4 100644 --- a/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct:free.toml +++ b/providers/openrouter/models/meta-llama/llama-3.3-70b-instruct:free.toml @@ -5,19 +5,19 @@ last_updated = "2024-12-06" attachment = false reasoning = false temperature = true -knowledge = "2024-12" tool_call = true -structured_output = true +structured_output = false +knowledge = "2023-12-31" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] -context = 131_072 +context = 65_536 output = 131_072 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-4-maverick.toml b/providers/openrouter/models/meta-llama/llama-4-maverick.toml new file mode 100644 index 000000000..78d9615bd --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-4-maverick.toml @@ -0,0 +1,23 @@ +name = "Llama 4 Maverick" +family = "llama" +release_date = "2025-04-05" +last_updated = "2025-04-05" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2024-08-31" +open_weights = true + +[cost] +input = 0.15 +output = 0.6 + +[limit] +context = 1_048_576 +output = 16_384 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-4-scout.toml b/providers/openrouter/models/meta-llama/llama-4-scout.toml new file mode 100644 index 000000000..6f34acaac --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-4-scout.toml @@ -0,0 +1,23 @@ +name = "Llama 4 Scout" +family = "llama" +release_date = "2025-04-05" +last_updated = "2025-04-05" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-08-31" +open_weights = true + +[cost] +input = 0.08 +output = 0.3 + +[limit] +context = 327_680 +output = 16_384 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-guard-3-8b.toml b/providers/openrouter/models/meta-llama/llama-guard-3-8b.toml new file mode 100644 index 000000000..c474f05d7 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-guard-3-8b.toml @@ -0,0 +1,23 @@ +name = "Llama Guard 3 8B" +family = "llama" +release_date = "2025-02-12" +last_updated = "2025-02-12" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.48 +output = 0.03 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/meta-llama/llama-guard-4-12b.toml b/providers/openrouter/models/meta-llama/llama-guard-4-12b.toml new file mode 100644 index 000000000..326cb1446 --- /dev/null +++ b/providers/openrouter/models/meta-llama/llama-guard-4-12b.toml @@ -0,0 +1,23 @@ +name = "Llama Guard 4 12B" +family = "llama" +release_date = "2025-04-30" +last_updated = "2025-04-30" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2024-08-31" +open_weights = true + +[cost] +input = 0.18 +output = 0.18 + +[limit] +context = 163_840 +output = 16_384 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/microsoft/phi-4-mini-instruct.toml b/providers/openrouter/models/microsoft/phi-4-mini-instruct.toml new file mode 100644 index 000000000..b73065978 --- /dev/null +++ b/providers/openrouter/models/microsoft/phi-4-mini-instruct.toml @@ -0,0 +1,23 @@ +name = "Phi 4 Mini Instruct" +family = "phi" +release_date = "2025-10-17" +last_updated = "2025-10-17" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +open_weights = true + +[cost] +input = 0.08 +output = 0.35 +cache_read = 0.08 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/microsoft/phi-4.toml b/providers/openrouter/models/microsoft/phi-4.toml new file mode 100644 index 000000000..a8a326c27 --- /dev/null +++ b/providers/openrouter/models/microsoft/phi-4.toml @@ -0,0 +1,23 @@ +name = "Phi 4" +family = "phi" +release_date = "2025-01-10" +last_updated = "2025-01-10" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2024-06-30" +open_weights = true + +[cost] +input = 0.065 +output = 0.14 + +[limit] +context = 16_384 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/microsoft/wizardlm-2-8x22b.toml b/providers/openrouter/models/microsoft/wizardlm-2-8x22b.toml new file mode 100644 index 000000000..2b859477b --- /dev/null +++ b/providers/openrouter/models/microsoft/wizardlm-2-8x22b.toml @@ -0,0 +1,23 @@ +name = "WizardLM-2 8x22B" +family = "o" +release_date = "2024-04-16" +last_updated = "2024-04-16" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2024-04-30" +open_weights = true + +[cost] +input = 0.62 +output = 0.62 + +[limit] +context = 65_535 +output = 8_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-01.toml b/providers/openrouter/models/minimax/minimax-01.toml index 8e330dbe1..d7a9b31fe 100644 --- a/providers/openrouter/models/minimax/minimax-01.toml +++ b/providers/openrouter/models/minimax/minimax-01.toml @@ -1,21 +1,23 @@ name = "MiniMax-01" family = "minimax" -attachment = true -reasoning = true -tool_call = true -temperature = true release_date = "2025-01-15" last_updated = "2025-01-15" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2024-03-31" open_weights = true [cost] -input = 0.20 -output = 1.10 +input = 0.2 +output = 1.1 [limit] -context = 1_000_000 -output = 1_000_000 +context = 1_000_192 +output = 1_000_192 [modalities] input = ["text", "image"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-m1.toml b/providers/openrouter/models/minimax/minimax-m1.toml index 18998a246..9e6c1964b 100644 --- a/providers/openrouter/models/minimax/minimax-m1.toml +++ b/providers/openrouter/models/minimax/minimax-m1.toml @@ -1,16 +1,18 @@ name = "MiniMax M1" family = "minimax" -attachment = false -reasoning = true -tool_call = true -temperature = true release_date = "2025-06-17" last_updated = "2025-06-17" -open_weights = true +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = false +knowledge = "2024-06-30" +open_weights = false [cost] -input = 0.40 -output = 2.20 +input = 0.4 +output = 2.2 [limit] context = 1_000_000 @@ -18,4 +20,4 @@ output = 40_000 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-m2-her.toml b/providers/openrouter/models/minimax/minimax-m2-her.toml new file mode 100644 index 000000000..8345ec56f --- /dev/null +++ b/providers/openrouter/models/minimax/minimax-m2-her.toml @@ -0,0 +1,23 @@ +name = "MiniMax M2-her" +family = "minimax" +release_date = "2026-01-23" +last_updated = "2026-01-23" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.03 + +[limit] +context = 65_536 +output = 2_048 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-m2.1.toml b/providers/openrouter/models/minimax/minimax-m2.1.toml index f2cee1a94..5a76251ea 100644 --- a/providers/openrouter/models/minimax/minimax-m2.1.toml +++ b/providers/openrouter/models/minimax/minimax-m2.1.toml @@ -1,25 +1,25 @@ name = "MiniMax M2.1" family = "minimax" -attachment = false -reasoning = true -tool_call = true -structured_output = true -temperature = true release_date = "2025-12-23" last_updated = "2025-12-23" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true open_weights = true [interleaved] field = "reasoning_details" [cost] -# Derived per-1k-token costs (approx) -input = 0.30 -output = 1.20 +input = 0.29 +output = 0.95 +cache_read = 0.03 [limit] -context = 204_800 -output = 131_072 +context = 196_608 +output = 196_608 [modalities] input = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-m2.5.toml b/providers/openrouter/models/minimax/minimax-m2.5.toml index 3076cdf06..46545c21d 100644 --- a/providers/openrouter/models/minimax/minimax-m2.5.toml +++ b/providers/openrouter/models/minimax/minimax-m2.5.toml @@ -1,25 +1,24 @@ name = "MiniMax M2.5" family = "minimax" -attachment = false -reasoning = true -tool_call = true -structured_output = true -temperature = true release_date = "2026-02-12" last_updated = "2026-02-12" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true open_weights = true [interleaved] field = "reasoning_details" [cost] -input = 0.30 -output = 1.20 -cache_read = 0.03 +input = 0.15 +output = 1.15 [limit] -context = 204_800 -output = 131_072 +context = 196_608 +output = 196_608 [modalities] input = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-m2.5:free.toml b/providers/openrouter/models/minimax/minimax-m2.5:free.toml index 659b06d65..a957239ce 100644 --- a/providers/openrouter/models/minimax/minimax-m2.5:free.toml +++ b/providers/openrouter/models/minimax/minimax-m2.5:free.toml @@ -1,24 +1,24 @@ name = "MiniMax M2.5 (free)" family = "minimax" -attachment = false -reasoning = true -tool_call = true -structured_output = true -temperature = true release_date = "2026-02-12" last_updated = "2026-02-12" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true open_weights = true [interleaved] field = "reasoning_details" [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] -context = 204_800 -output = 131_072 +context = 196_608 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/openrouter/models/minimax/minimax-m2.7.toml b/providers/openrouter/models/minimax/minimax-m2.7.toml index cef5825f6..25804b41d 100644 --- a/providers/openrouter/models/minimax/minimax-m2.7.toml +++ b/providers/openrouter/models/minimax/minimax-m2.7.toml @@ -6,16 +6,15 @@ attachment = false reasoning = true temperature = true tool_call = true +structured_output = true open_weights = true [cost] -input = 0.30 -output = 1.20 -cache_read = 0.06 -cache_write = 0.375 +input = 0.279 +output = 1.2 [limit] -context = 204_800 +context = 196_608 output = 131_072 [modalities] diff --git a/providers/openrouter/models/minimax/minimax-m2.toml b/providers/openrouter/models/minimax/minimax-m2.toml index f9c9c7ad9..e364265a3 100644 --- a/providers/openrouter/models/minimax/minimax-m2.toml +++ b/providers/openrouter/models/minimax/minimax-m2.toml @@ -1,26 +1,25 @@ name = "MiniMax M2" family = "minimax" -attachment = false -reasoning = true -tool_call = true -structured_output = true -temperature = true release_date = "2025-10-23" last_updated = "2025-10-23" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true open_weights = true [interleaved] field = "reasoning_details" [cost] -input = 0.28 -output = 1.15 -cache_read = 0.28 -cache_write = 1.15 +input = 0.255 +output = 1 +cache_read = 0.03 [limit] -context = 196_600 -output = 118_000 +context = 196_608 +output = 196_608 [modalities] input = ["text"] diff --git a/providers/openrouter/models/mistralai/codestral-2508.toml b/providers/openrouter/models/mistralai/codestral-2508.toml index b3f0f6e45..acd5cea96 100644 --- a/providers/openrouter/models/mistralai/codestral-2508.toml +++ b/providers/openrouter/models/mistralai/codestral-2508.toml @@ -2,22 +2,23 @@ name = "Codestral 2508" family = "codestral" release_date = "2025-08-01" last_updated = "2025-08-01" -attachment = false +attachment = true reasoning = false temperature = true -knowledge = "2025-05" tool_call = true structured_output = true -open_weights = true +knowledge = "2025-03-31" +open_weights = false [cost] -input = 0.30 -output = 0.90 +input = 0.3 +output = 0.9 +cache_read = 0.03 [limit] -context = 256_000 +context = 256_000 output = 256_000 [modalities] -input = ["text"] -output = ["text"] \ No newline at end of file +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/devstral-2512.toml b/providers/openrouter/models/mistralai/devstral-2512.toml index ce9ea7fae..cf601f1f4 100644 --- a/providers/openrouter/models/mistralai/devstral-2512.toml +++ b/providers/openrouter/models/mistralai/devstral-2512.toml @@ -1,23 +1,24 @@ name = "Devstral 2 2512" family = "devstral" -release_date = "2025-09-12" -last_updated = "2025-09-12" -attachment = false +release_date = "2025-12-09" +last_updated = "2025-12-09" +attachment = true reasoning = false temperature = true -knowledge = "2025-12" tool_call = true structured_output = true +knowledge = "2025-12" open_weights = true [cost] -input = 0.15 -output = 0.60 +input = 0.4 +output = 2 +cache_read = 0.04 [limit] context = 262_144 output = 262_144 [modalities] -input = ["text"] +input = ["text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/mistralai/devstral-medium-2507.toml b/providers/openrouter/models/mistralai/devstral-medium.toml similarity index 58% rename from providers/openrouter/models/mistralai/devstral-medium-2507.toml rename to providers/openrouter/models/mistralai/devstral-medium.toml index 6b890e44e..6d951fa76 100644 --- a/providers/openrouter/models/mistralai/devstral-medium-2507.toml +++ b/providers/openrouter/models/mistralai/devstral-medium.toml @@ -2,22 +2,23 @@ name = "Devstral Medium" family = "devstral" release_date = "2025-07-10" last_updated = "2025-07-10" -attachment = false +attachment = true reasoning = false temperature = true -knowledge = "2025-05" tool_call = true structured_output = true -open_weights = true +knowledge = "2025-06-30" +open_weights = false [cost] -input = 0.40 -output = 2.00 +input = 0.4 +output = 2 +cache_read = 0.04 [limit] -context = 131_072 +context = 131_072 output = 131_072 [modalities] -input = ["text"] -output = ["text"] \ No newline at end of file +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/devstral-small-2505.toml b/providers/openrouter/models/mistralai/devstral-small-2505.toml deleted file mode 100644 index b30a5bfc9..000000000 --- a/providers/openrouter/models/mistralai/devstral-small-2505.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Devstral Small" -family = "devstral" -release_date = "2025-05-07" -last_updated = "2025-05-07" -attachment = false -reasoning = false -temperature = true -knowledge = "2025-05" -tool_call = true -open_weights = true - -[cost] -input = 0.06 -output = 0.12 - -[limit] -context = 128_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/openrouter/models/mistralai/devstral-small-2507.toml b/providers/openrouter/models/mistralai/devstral-small.toml similarity index 67% rename from providers/openrouter/models/mistralai/devstral-small-2507.toml rename to providers/openrouter/models/mistralai/devstral-small.toml index a44dc68e1..30e54299e 100644 --- a/providers/openrouter/models/mistralai/devstral-small-2507.toml +++ b/providers/openrouter/models/mistralai/devstral-small.toml @@ -2,22 +2,23 @@ name = "Devstral Small 1.1" family = "devstral" release_date = "2025-07-10" last_updated = "2025-07-10" -attachment = false +attachment = true reasoning = false temperature = true -knowledge = "2025-05" tool_call = true structured_output = true +knowledge = "2025-03-31" open_weights = true [cost] -input = 0.10 -output = 0.30 +input = 0.1 +output = 0.3 +cache_read = 0.01 [limit] -context = 131_072 +context = 131_072 output = 131_072 [modalities] -input = ["text"] +input = ["text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/mistralai/ministral-14b-2512.toml b/providers/openrouter/models/mistralai/ministral-14b-2512.toml new file mode 100644 index 000000000..aa4d31f40 --- /dev/null +++ b/providers/openrouter/models/mistralai/ministral-14b-2512.toml @@ -0,0 +1,23 @@ +name = "Ministral 3 14B 2512" +family = "ministral" +release_date = "2025-12-02" +last_updated = "2025-12-02" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.2 +output = 0.2 +cache_read = 0.02 + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/ministral-3b-2512.toml b/providers/openrouter/models/mistralai/ministral-3b-2512.toml new file mode 100644 index 000000000..93588874e --- /dev/null +++ b/providers/openrouter/models/mistralai/ministral-3b-2512.toml @@ -0,0 +1,23 @@ +name = "Ministral 3 3B 2512" +family = "ministral" +release_date = "2025-12-02" +last_updated = "2025-12-02" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.1 +output = 0.1 +cache_read = 0.01 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/ministral-8b-2512.toml b/providers/openrouter/models/mistralai/ministral-8b-2512.toml new file mode 100644 index 000000000..c18ff718c --- /dev/null +++ b/providers/openrouter/models/mistralai/ministral-8b-2512.toml @@ -0,0 +1,23 @@ +name = "Ministral 3 8B 2512" +family = "ministral" +release_date = "2025-12-02" +last_updated = "2025-12-02" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.15 +output = 0.15 +cache_read = 0.015 + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-7b-instruct-v0.1.toml b/providers/openrouter/models/mistralai/mistral-7b-instruct-v0.1.toml new file mode 100644 index 000000000..f1547cd44 --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-7b-instruct-v0.1.toml @@ -0,0 +1,23 @@ +name = "Mistral 7B Instruct v0.1" +family = "mistral" +release_date = "2023-09-28" +last_updated = "2023-09-28" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-09-30" +open_weights = true + +[cost] +input = 0.11 +output = 0.19 + +[limit] +context = 2_824 +output = 2_824 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-large-2407.toml b/providers/openrouter/models/mistralai/mistral-large-2407.toml new file mode 100644 index 000000000..fd29d1702 --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-large-2407.toml @@ -0,0 +1,24 @@ +name = "Mistral Large 2407" +family = "mistral-large" +release_date = "2024-11-19" +last_updated = "2024-11-19" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-03-31" +open_weights = false + +[cost] +input = 2 +output = 6 +cache_read = 0.2 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-large-2411.toml b/providers/openrouter/models/mistralai/mistral-large-2411.toml new file mode 100644 index 000000000..2dc0b1be5 --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-large-2411.toml @@ -0,0 +1,24 @@ +name = "Mistral Large 2411" +family = "mistral-large" +release_date = "2024-11-19" +last_updated = "2024-11-19" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-07-31" +open_weights = false + +[cost] +input = 2 +output = 6 +cache_read = 0.2 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-large-2512.toml b/providers/openrouter/models/mistralai/mistral-large-2512.toml new file mode 100644 index 000000000..763b5a499 --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-large-2512.toml @@ -0,0 +1,23 @@ +name = "Mistral Large 3 2512" +family = "mistral-large" +release_date = "2025-12-01" +last_updated = "2025-12-01" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.5 +output = 1.5 +cache_read = 0.05 + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-large.toml b/providers/openrouter/models/mistralai/mistral-large.toml new file mode 100644 index 000000000..eaf17854c --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-large.toml @@ -0,0 +1,24 @@ +name = "Mistral Large" +family = "mistral-large" +release_date = "2024-02-26" +last_updated = "2024-02-26" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-11-30" +open_weights = false + +[cost] +input = 2 +output = 6 +cache_read = 0.2 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-medium-3-5.toml b/providers/openrouter/models/mistralai/mistral-medium-3-5.toml new file mode 100644 index 000000000..76fdb4df9 --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-medium-3-5.toml @@ -0,0 +1,22 @@ +name = "Mistral Medium 3.5" +family = "mistral-medium" +release_date = "2026-04-30" +last_updated = "2026-04-30" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 1.5 +output = 7.5 + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-medium-3.1.toml b/providers/openrouter/models/mistralai/mistral-medium-3.1.toml index e9412dec6..5620a6fad 100644 --- a/providers/openrouter/models/mistralai/mistral-medium-3.1.toml +++ b/providers/openrouter/models/mistralai/mistral-medium-3.1.toml @@ -1,23 +1,24 @@ name = "Mistral Medium 3.1" family = "mistral-medium" -release_date = "2025-08-12" -last_updated = "2025-08-12" +release_date = "2025-08-13" +last_updated = "2025-08-13" attachment = true reasoning = false temperature = true -knowledge = "2025-05" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = false [cost] -input = 0.40 -output = 2.00 +input = 0.4 +output = 2 +cache_read = 0.04 [limit] -context = 262_144 +context = 131_072 output = 262_144 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-medium-3.toml b/providers/openrouter/models/mistralai/mistral-medium-3.toml index f11e24b81..2dbb9e88f 100644 --- a/providers/openrouter/models/mistralai/mistral-medium-3.toml +++ b/providers/openrouter/models/mistralai/mistral-medium-3.toml @@ -5,19 +5,20 @@ last_updated = "2025-05-07" attachment = true reasoning = false temperature = true -knowledge = "2025-05" tool_call = true structured_output = true +knowledge = "2025-03-31" open_weights = false [cost] -input = 0.40 -output = 2.00 +input = 0.4 +output = 2 +cache_read = 0.04 [limit] context = 131_072 output = 131_072 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-nemo.toml b/providers/openrouter/models/mistralai/mistral-nemo.toml new file mode 100644 index 000000000..3357e12ce --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-nemo.toml @@ -0,0 +1,23 @@ +name = "Mistral Nemo" +family = "mistral-nemo" +release_date = "2024-07-19" +last_updated = "2024-07-19" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-04-30" +open_weights = true + +[cost] +input = 0.02 +output = 0.03 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-saba.toml b/providers/openrouter/models/mistralai/mistral-saba.toml new file mode 100644 index 000000000..4a5b8bca3 --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-saba.toml @@ -0,0 +1,24 @@ +name = "Saba" +family = "mistral" +release_date = "2025-02-17" +last_updated = "2025-02-17" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-09-30" +open_weights = false + +[cost] +input = 0.2 +output = 0.6 +cache_read = 0.02 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-small-24b-instruct-2501.toml b/providers/openrouter/models/mistralai/mistral-small-24b-instruct-2501.toml new file mode 100644 index 000000000..83aa6e83d --- /dev/null +++ b/providers/openrouter/models/mistralai/mistral-small-24b-instruct-2501.toml @@ -0,0 +1,23 @@ +name = "Mistral Small 3" +family = "mistral-small" +release_date = "2025-01-30" +last_updated = "2025-01-30" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-10-31" +open_weights = true + +[cost] +input = 0.05 +output = 0.08 + +[limit] +context = 32_768 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-small-2603.toml b/providers/openrouter/models/mistralai/mistral-small-2603.toml index dac441994..24518c209 100644 --- a/providers/openrouter/models/mistralai/mistral-small-2603.toml +++ b/providers/openrouter/models/mistralai/mistral-small-2603.toml @@ -1,4 +1,3 @@ -id = "mistralai/mistral-small-2603" name = "Mistral Small 4" family = "mistral-small" release_date = "2026-03-16" @@ -6,13 +5,15 @@ last_updated = "2026-03-16" attachment = true reasoning = true temperature = true -knowledge = "2025-06" tool_call = true +structured_output = true +knowledge = "2025-06" open_weights = true [cost] input = 0.15 -output = 0.60 +output = 0.6 +cache_read = 0.015 [limit] context = 262_144 diff --git a/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml index 85cd3d677..45c663140 100644 --- a/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml @@ -1,18 +1,23 @@ -id = "mistralai/mistral-small-3.1-24b-instruct:free" -name = "Mistral Small 3.1 24B Instruct" +name = "Mistral Small 3.1 24B" family = "mistral-small" release_date = "2025-03-17" last_updated = "2025-03-17" attachment = true reasoning = false temperature = true -knowledge = "2024-10" -tool_call = true -structured_output = true +tool_call = false +structured_output = false +knowledge = "2023-10-31" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 128000, output = 8192 } + +[cost] +input = 0.35 +output = 0.56 + +[limit] +context = 128_000 +output = 8_192 [modalities] input = ["text", "image"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mistral-small-3.2-24b-instruct.toml b/providers/openrouter/models/mistralai/mistral-small-3.2-24b-instruct.toml index 497ece4f6..dc7911006 100644 --- a/providers/openrouter/models/mistralai/mistral-small-3.2-24b-instruct.toml +++ b/providers/openrouter/models/mistralai/mistral-small-3.2-24b-instruct.toml @@ -1,18 +1,23 @@ -id = "mistralai/mistral-small-3.2-24b-instruct:free" -name = "Mistral Small 3.2 24B Instruct" +name = "Mistral Small 3.2 24B" family = "mistral-small" release_date = "2025-06-20" last_updated = "2025-06-20" attachment = true reasoning = false temperature = true -knowledge = "2024-10" tool_call = true structured_output = true +knowledge = "2023-10-31" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 96000, output = 8192 } + +[cost] +input = 0.075 +output = 0.2 + +[limit] +context = 128_000 +output = 16_384 [modalities] -input = ["text", "image"] -output = ["text"] \ No newline at end of file +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/mixtral-8x22b-instruct.toml b/providers/openrouter/models/mistralai/mixtral-8x22b-instruct.toml new file mode 100644 index 000000000..9e6a7c7dd --- /dev/null +++ b/providers/openrouter/models/mistralai/mixtral-8x22b-instruct.toml @@ -0,0 +1,24 @@ +name = "Mixtral 8x22B Instruct" +family = "mistral" +release_date = "2024-04-17" +last_updated = "2024-04-17" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-01-31" +open_weights = true + +[cost] +input = 2 +output = 6 +cache_read = 0.2 + +[limit] +context = 65_536 +output = 65_536 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/pixtral-large-2411.toml b/providers/openrouter/models/mistralai/pixtral-large-2411.toml new file mode 100644 index 000000000..17e6155cb --- /dev/null +++ b/providers/openrouter/models/mistralai/pixtral-large-2411.toml @@ -0,0 +1,24 @@ +name = "Pixtral Large 2411" +family = "mistral" +release_date = "2024-11-19" +last_updated = "2024-11-19" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-07-31" +open_weights = false + +[cost] +input = 2 +output = 6 +cache_read = 0.2 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/mistralai/voxtral-small-24b-2507.toml b/providers/openrouter/models/mistralai/voxtral-small-24b-2507.toml new file mode 100644 index 000000000..9fc69d1db --- /dev/null +++ b/providers/openrouter/models/mistralai/voxtral-small-24b-2507.toml @@ -0,0 +1,23 @@ +name = "Voxtral Small 24B 2507" +family = "mistral" +release_date = "2025-10-30" +last_updated = "2025-10-30" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.1 +output = 0.3 +cache_read = 0.01 + +[limit] +context = 32_000 +output = 32_000 + +[modalities] +input = ["text", "audio", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/moonshotai/kimi-k2-0905.toml b/providers/openrouter/models/moonshotai/kimi-k2-0905.toml index da423e284..fc5dbdc3e 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2-0905.toml +++ b/providers/openrouter/models/moonshotai/kimi-k2-0905.toml @@ -1,13 +1,13 @@ -name = "Kimi K2 Instruct 0905" +name = "Kimi K2 0905" family = "kimi" -release_date = "2025-09-05" -last_updated = "2025-09-05" +release_date = "2025-09-04" +last_updated = "2025-09-04" attachment = false reasoning = false temperature = true tool_call = true structured_output = true -knowledge = "2024-10" +knowledge = "2024-12-31" open_weights = true [cost] @@ -16,8 +16,8 @@ output = 2.5 [limit] context = 262_144 -output = 16_384 +output = 262_144 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/moonshotai/kimi-k2.5.toml b/providers/openrouter/models/moonshotai/kimi-k2.5.toml index 24c6a86d0..83c3cbbf3 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2.5.toml +++ b/providers/openrouter/models/moonshotai/kimi-k2.5.toml @@ -14,14 +14,14 @@ open_weights = true field = "reasoning_details" [cost] -input = 0.60 -output = 3.00 -cache_read = 0.10 +input = 0.4 +output = 1.9 +cache_read = 0.09 [limit] context = 262_144 output = 262_144 [modalities] -input = ["text", "image", "video"] +input = ["text", "image"] output = ["text"] diff --git a/providers/openrouter/models/moonshotai/kimi-k2.6.toml b/providers/openrouter/models/moonshotai/kimi-k2.6.toml index 2c2370f5c..b774e1d08 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2.6.toml +++ b/providers/openrouter/models/moonshotai/kimi-k2.6.toml @@ -13,13 +13,13 @@ open_weights = true field = "reasoning_details" [cost] -input = 0.95 -output = 4.00 -cache_read = 0.16 +input = 0.73 +output = 3.49 +cache_read = 0.25 [limit] -context = 262_144 -output = 262_144 +context = 262_142 +output = 262_142 [modalities] input = ["text", "image"] diff --git a/providers/openrouter/models/moonshotai/kimi-k2.toml b/providers/openrouter/models/moonshotai/kimi-k2.toml index f8e2c25c6..4c2619b41 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2.toml +++ b/providers/openrouter/models/moonshotai/kimi-k2.toml @@ -1,4 +1,4 @@ -name = "Kimi K2" +name = "Kimi K2 0711" family = "kimi" release_date = "2025-07-11" last_updated = "2025-07-11" @@ -6,12 +6,13 @@ attachment = false reasoning = false temperature = true tool_call = true -knowledge = "2024-10" +structured_output = false +knowledge = "2024-12-31" open_weights = true [cost] -input = 0.55 -output = 2.20 +input = 0.57 +output = 2.3 [limit] context = 131_072 diff --git a/providers/openrouter/models/morph/morph-v3-fast.toml b/providers/openrouter/models/morph/morph-v3-fast.toml new file mode 100644 index 000000000..886d5c1ef --- /dev/null +++ b/providers/openrouter/models/morph/morph-v3-fast.toml @@ -0,0 +1,22 @@ +name = "Morph V3 Fast" +family = "morph" +release_date = "2025-07-07" +last_updated = "2025-07-07" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.8 +output = 1.2 + +[limit] +context = 81_920 +output = 38_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/morph/morph-v3-large.toml b/providers/openrouter/models/morph/morph-v3-large.toml new file mode 100644 index 000000000..9ff971031 --- /dev/null +++ b/providers/openrouter/models/morph/morph-v3-large.toml @@ -0,0 +1,22 @@ +name = "Morph V3 Large" +family = "morph" +release_date = "2025-07-07" +last_updated = "2025-07-07" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.9 +output = 1.9 + +[limit] +context = 262_144 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nex-agi/deepseek-v3.1-nex-n1.toml b/providers/openrouter/models/nex-agi/deepseek-v3.1-nex-n1.toml new file mode 100644 index 000000000..14f5fb760 --- /dev/null +++ b/providers/openrouter/models/nex-agi/deepseek-v3.1-nex-n1.toml @@ -0,0 +1,22 @@ +name = "DeepSeek V3.1 Nex N1" +family = "deepseek" +release_date = "2025-12-08" +last_updated = "2025-12-08" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.135 +output = 0.5 + +[limit] +context = 131_072 +output = 163_840 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nousresearch/hermes-2-pro-llama-3-8b.toml b/providers/openrouter/models/nousresearch/hermes-2-pro-llama-3-8b.toml new file mode 100644 index 000000000..69637ffdb --- /dev/null +++ b/providers/openrouter/models/nousresearch/hermes-2-pro-llama-3-8b.toml @@ -0,0 +1,23 @@ +name = "Hermes 2 Pro - Llama-3 8B" +family = "nousresearch" +release_date = "2024-05-27" +last_updated = "2024-05-27" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.14 +output = 0.14 + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b.toml b/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b.toml new file mode 100644 index 000000000..151c86aa4 --- /dev/null +++ b/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b.toml @@ -0,0 +1,23 @@ +name = "Hermes 3 405B Instruct" +family = "nousresearch" +release_date = "2024-08-16" +last_updated = "2024-08-16" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 1 +output = 1 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b:free.toml b/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b:free.toml index 150a08de5..484674831 100644 --- a/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b:free.toml +++ b/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-405b:free.toml @@ -3,15 +3,16 @@ family = "hermes" release_date = "2024-08-16" last_updated = "2024-08-16" attachment = false -reasoning = true +reasoning = false temperature = true -knowledge = "2023-12" tool_call = false +structured_output = false +knowledge = "2023-12-31" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 131_072 @@ -19,4 +20,4 @@ output = 131_072 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-70b.toml b/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-70b.toml new file mode 100644 index 000000000..eccbcec7d --- /dev/null +++ b/providers/openrouter/models/nousresearch/hermes-3-llama-3.1-70b.toml @@ -0,0 +1,23 @@ +name = "Hermes 3 70B Instruct" +family = "nousresearch" +release_date = "2024-08-18" +last_updated = "2024-08-18" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.3 +output = 0.3 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nousresearch/hermes-4-405b.toml b/providers/openrouter/models/nousresearch/hermes-4-405b.toml index 5a5b103e1..fd49ff433 100644 --- a/providers/openrouter/models/nousresearch/hermes-4-405b.toml +++ b/providers/openrouter/models/nousresearch/hermes-4-405b.toml @@ -1,18 +1,18 @@ -id = "nousresearch/hermes-4-405b" name = "Hermes 4 405B" family = "hermes" -release_date = "2025-08-25" -last_updated = "2025-08-25" +release_date = "2025-08-26" +last_updated = "2025-08-26" attachment = false reasoning = true temperature = true -knowledge = "2023-12" -tool_call = true +tool_call = false +structured_output = true +knowledge = "2024-08-31" open_weights = true [cost] -input = 1.00 -output = 3.00 +input = 1 +output = 3 [limit] context = 131_072 @@ -20,4 +20,4 @@ output = 131_072 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/nousresearch/hermes-4-70b.toml b/providers/openrouter/models/nousresearch/hermes-4-70b.toml index b75866907..771354c9e 100644 --- a/providers/openrouter/models/nousresearch/hermes-4-70b.toml +++ b/providers/openrouter/models/nousresearch/hermes-4-70b.toml @@ -1,19 +1,18 @@ -id = "nousresearch/hermes-4-405b" name = "Hermes 4 70B" family = "hermes" -release_date = "2025-08-25" -last_updated = "2025-08-25" +release_date = "2025-08-26" +last_updated = "2025-08-26" attachment = false reasoning = true temperature = true -knowledge = "2023-12" -tool_call = true +tool_call = false structured_output = true +knowledge = "2024-08-31" open_weights = true [cost] input = 0.13 -output = 0.40 +output = 0.4 [limit] context = 131_072 @@ -21,4 +20,4 @@ output = 131_072 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/nvidia/llama-3.3-nemotron-super-49b-v1.5.toml b/providers/openrouter/models/nvidia/llama-3.3-nemotron-super-49b-v1.5.toml new file mode 100644 index 000000000..49a3655ba --- /dev/null +++ b/providers/openrouter/models/nvidia/llama-3.3-nemotron-super-49b-v1.5.toml @@ -0,0 +1,23 @@ +name = "Llama 3.3 Nemotron Super 49B V1.5" +family = "nemotron" +release_date = "2025-10-10" +last_updated = "2025-10-10" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-03-31" +open_weights = true + +[cost] +input = 0.1 +output = 0.4 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml new file mode 100644 index 000000000..3d57b328d --- /dev/null +++ b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml @@ -0,0 +1,22 @@ +name = "Nemotron 3 Nano 30B A3B" +family = "nemotron" +release_date = "2025-12-14" +last_updated = "2025-12-14" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.05 +output = 0.2 + +[limit] +context = 262_144 +output = 228_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b:free.toml b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b:free.toml index b049a8441..31f571ae6 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b:free.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b:free.toml @@ -1,19 +1,18 @@ name = "Nemotron 3 Nano 30B A3B (free)" family = "nemotron" - release_date = "2025-12-14" -last_updated = "2026-01-31" +last_updated = "2025-12-14" attachment = false reasoning = true temperature = true -knowledge = "2025-11" tool_call = true -structured_output = true +structured_output = false +knowledge = "2025-11" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 256_000 diff --git a/providers/openrouter/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free.toml b/providers/openrouter/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free.toml index f9c0db4b9..5fcd435a7 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free.toml @@ -1,23 +1,22 @@ name = "Nemotron 3 Nano Omni (free)" family = "nemotron" - release_date = "2026-04-28" last_updated = "2026-04-28" attachment = true reasoning = true temperature = true tool_call = true -structured_output = true -open_weights = true +structured_output = false +open_weights = false [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 256_000 output = 65_536 [modalities] -input = ["text", "image", "video", "audio"] +input = ["text", "audio", "image", "video"] output = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b.toml b/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b.toml index bbcdfcd30..30cea482f 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b.toml @@ -1,22 +1,23 @@ name = "Nemotron 3 Super" family = "nemotron" -attachment = false -reasoning = true -tool_call = true -temperature = true -knowledge = "2024-04" release_date = "2026-03-11" last_updated = "2026-03-11" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-04" open_weights = true [cost] -input = 0.10 -output = 0.50 +input = 0.09 +output = 0.45 [limit] -context = 262144 -output = 262144 +context = 262_144 +output = 262_144 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b:free.toml b/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b:free.toml index 528a20a35..98caeb0e7 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b:free.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-super-120b-a12b:free.toml @@ -1,21 +1,22 @@ name = "Nemotron 3 Super (free)" family = "nemotron" -attachment = false -reasoning = true -tool_call = true -temperature = true -knowledge = "2024-04" release_date = "2026-03-11" last_updated = "2026-03-11" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-04" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] -context = 262144 -output = 262144 +context = 262_144 +output = 262_144 [modalities] input = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-nano-12b-v2-vl:free.toml b/providers/openrouter/models/nvidia/nemotron-nano-12b-v2-vl:free.toml index 48cc922a6..2dc8afc71 100644 --- a/providers/openrouter/models/nvidia/nemotron-nano-12b-v2-vl:free.toml +++ b/providers/openrouter/models/nvidia/nemotron-nano-12b-v2-vl:free.toml @@ -1,23 +1,23 @@ name = "Nemotron Nano 12B 2 VL (free)" family = "nemotron" - release_date = "2025-10-28" -last_updated = "2026-01-31" -attachment = false +last_updated = "2025-10-28" +attachment = true reasoning = true temperature = true -knowledge = "2025-11" tool_call = true +structured_output = false +knowledge = "2025-11" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 128_000 output = 128_000 [modalities] -input = ["text","image"] +input = ["image", "text", "video"] output = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-nano-9b-v2.toml b/providers/openrouter/models/nvidia/nemotron-nano-9b-v2.toml index 6eb62ba43..e77d67bef 100644 --- a/providers/openrouter/models/nvidia/nemotron-nano-9b-v2.toml +++ b/providers/openrouter/models/nvidia/nemotron-nano-9b-v2.toml @@ -1,13 +1,13 @@ -name = "nvidia-nemotron-nano-9b-v2" +name = "Nemotron Nano 9B V2" family = "nemotron" - -release_date = "2025-08-18" -last_updated = "2025-08-18" +release_date = "2025-09-05" +last_updated = "2025-09-05" attachment = false reasoning = true temperature = true -knowledge = "2024-09" tool_call = true +structured_output = true +knowledge = "2025-03-31" open_weights = true [cost] @@ -16,7 +16,7 @@ output = 0.16 [limit] context = 131_072 -output = 131_072 +output = 16_384 [modalities] input = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-nano-9b-v2:free.toml b/providers/openrouter/models/nvidia/nemotron-nano-9b-v2:free.toml index 0ae490984..dca3f3c8d 100644 --- a/providers/openrouter/models/nvidia/nemotron-nano-9b-v2:free.toml +++ b/providers/openrouter/models/nvidia/nemotron-nano-9b-v2:free.toml @@ -1,19 +1,18 @@ name = "Nemotron Nano 9B V2 (free)" family = "nemotron" - release_date = "2025-09-05" -last_updated = "2025-08-18" +last_updated = "2025-09-05" attachment = false reasoning = true temperature = true -knowledge = "2024-09" tool_call = true structured_output = true +knowledge = "2025-03-31" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 128_000 diff --git a/providers/openrouter/models/openai/gpt-3.5-turbo-0613.toml b/providers/openrouter/models/openai/gpt-3.5-turbo-0613.toml new file mode 100644 index 000000000..78d97616b --- /dev/null +++ b/providers/openrouter/models/openai/gpt-3.5-turbo-0613.toml @@ -0,0 +1,23 @@ +name = "GPT-3.5 Turbo (older v0613)" +family = "gpt" +release_date = "2024-01-25" +last_updated = "2024-01-25" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2021-09-30" +open_weights = false + +[cost] +input = 1 +output = 2 + +[limit] +context = 4_095 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-3.5-turbo-16k.toml b/providers/openrouter/models/openai/gpt-3.5-turbo-16k.toml new file mode 100644 index 000000000..082e15878 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-3.5-turbo-16k.toml @@ -0,0 +1,23 @@ +name = "GPT-3.5 Turbo 16k" +family = "gpt" +release_date = "2023-08-28" +last_updated = "2023-08-28" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2021-09-30" +open_weights = false + +[cost] +input = 3 +output = 4 + +[limit] +context = 16_385 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-3.5-turbo-instruct.toml b/providers/openrouter/models/openai/gpt-3.5-turbo-instruct.toml new file mode 100644 index 000000000..c4bb228e8 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-3.5-turbo-instruct.toml @@ -0,0 +1,23 @@ +name = "GPT-3.5 Turbo Instruct" +family = "gpt" +release_date = "2023-09-28" +last_updated = "2023-09-28" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2021-09-30" +open_weights = false + +[cost] +input = 1.5 +output = 2 + +[limit] +context = 4_095 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-3.5-turbo.toml b/providers/openrouter/models/openai/gpt-3.5-turbo.toml new file mode 100644 index 000000000..8c0d70235 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-3.5-turbo.toml @@ -0,0 +1,23 @@ +name = "GPT-3.5 Turbo" +family = "gpt" +release_date = "2023-05-28" +last_updated = "2023-05-28" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2021-09-30" +open_weights = false + +[cost] +input = 0.5 +output = 1.5 + +[limit] +context = 16_385 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4-0314.toml b/providers/openrouter/models/openai/gpt-4-0314.toml new file mode 100644 index 000000000..8042181af --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4-0314.toml @@ -0,0 +1,23 @@ +name = "GPT-4 (older v0314)" +family = "gpt" +release_date = "2023-05-28" +last_updated = "2023-05-28" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2021-09-30" +open_weights = false + +[cost] +input = 30 +output = 60 + +[limit] +context = 8_191 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4-1106-preview.toml b/providers/openrouter/models/openai/gpt-4-1106-preview.toml new file mode 100644 index 000000000..3d6a5c5a1 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4-1106-preview.toml @@ -0,0 +1,23 @@ +name = "GPT-4 Turbo (older v1106)" +family = "gpt" +release_date = "2023-11-06" +last_updated = "2023-11-06" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-04-30" +open_weights = false + +[cost] +input = 10 +output = 30 + +[limit] +context = 128_000 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4-turbo-preview.toml b/providers/openrouter/models/openai/gpt-4-turbo-preview.toml new file mode 100644 index 000000000..f91511d9b --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4-turbo-preview.toml @@ -0,0 +1,23 @@ +name = "GPT-4 Turbo Preview" +family = "gpt" +release_date = "2024-01-25" +last_updated = "2024-01-25" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-12-31" +open_weights = false + +[cost] +input = 10 +output = 30 + +[limit] +context = 128_000 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4-turbo.toml b/providers/openrouter/models/openai/gpt-4-turbo.toml new file mode 100644 index 000000000..9d4ad7ef0 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4-turbo.toml @@ -0,0 +1,23 @@ +name = "GPT-4 Turbo" +family = "gpt" +release_date = "2024-04-09" +last_updated = "2024-04-09" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-12-31" +open_weights = false + +[cost] +input = 10 +output = 30 + +[limit] +context = 128_000 +output = 4_096 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4.1-mini.toml b/providers/openrouter/models/openai/gpt-4.1-mini.toml index b42c69a16..c66d2f71d 100644 --- a/providers/openrouter/models/openai/gpt-4.1-mini.toml +++ b/providers/openrouter/models/openai/gpt-4.1-mini.toml @@ -7,18 +7,18 @@ reasoning = false temperature = true tool_call = true structured_output = true -knowledge = "2024-04" +knowledge = "2024-06-30" open_weights = false [cost] -input = 0.40 -output = 1.60 -cache_read = 0.10 +input = 0.4 +output = 1.6 +cache_read = 0.1 [limit] context = 1_047_576 output = 32_768 [modalities] -input = ["text", "image"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4.1-nano.toml b/providers/openrouter/models/openai/gpt-4.1-nano.toml new file mode 100644 index 000000000..af841e52e --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4.1-nano.toml @@ -0,0 +1,24 @@ +name = "GPT-4.1 Nano" +family = "gpt" +release_date = "2025-04-14" +last_updated = "2025-04-14" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-06-30" +open_weights = false + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.025 + +[limit] +context = 1_047_576 +output = 32_768 + +[modalities] +input = ["image", "text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4.1.toml b/providers/openrouter/models/openai/gpt-4.1.toml index 2d25d2131..7d8a9c33f 100644 --- a/providers/openrouter/models/openai/gpt-4.1.toml +++ b/providers/openrouter/models/openai/gpt-4.1.toml @@ -7,18 +7,18 @@ reasoning = false temperature = true tool_call = true structured_output = true -knowledge = "2024-04" +knowledge = "2024-06-30" open_weights = false [cost] -input = 2.00 -output = 8.00 -cache_read = 0.50 +input = 2 +output = 8 +cache_read = 0.5 [limit] context = 1_047_576 output = 32_768 [modalities] -input = ["text", "image"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4.toml b/providers/openrouter/models/openai/gpt-4.toml new file mode 100644 index 000000000..fc06d7104 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4.toml @@ -0,0 +1,23 @@ +name = "GPT-4" +family = "gpt" +release_date = "2023-05-28" +last_updated = "2023-05-28" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2021-09-30" +open_weights = false + +[cost] +input = 30 +output = 60 + +[limit] +context = 8_191 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-2024-05-13.toml b/providers/openrouter/models/openai/gpt-4o-2024-05-13.toml new file mode 100644 index 000000000..14cfc095b --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-2024-05-13.toml @@ -0,0 +1,23 @@ +name = "GPT-4o (2024-05-13)" +family = "gpt" +release_date = "2024-05-13" +last_updated = "2024-05-13" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 5 +output = 15 + +[limit] +context = 128_000 +output = 4_096 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-2024-08-06.toml b/providers/openrouter/models/openai/gpt-4o-2024-08-06.toml new file mode 100644 index 000000000..58ad1de85 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-2024-08-06.toml @@ -0,0 +1,24 @@ +name = "GPT-4o (2024-08-06)" +family = "gpt" +release_date = "2024-08-06" +last_updated = "2024-08-06" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-2024-11-20.toml b/providers/openrouter/models/openai/gpt-4o-2024-11-20.toml new file mode 100644 index 000000000..fe45f89ea --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-2024-11-20.toml @@ -0,0 +1,24 @@ +name = "GPT-4o (2024-11-20)" +family = "gpt" +release_date = "2024-11-20" +last_updated = "2024-11-20" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-audio-preview.toml b/providers/openrouter/models/openai/gpt-4o-audio-preview.toml new file mode 100644 index 000000000..9420531f4 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-audio-preview.toml @@ -0,0 +1,23 @@ +name = "GPT-4o Audio" +family = "gpt" +release_date = "2025-08-15" +last_updated = "2025-08-15" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["audio", "text"] +output = ["text", "audio"] diff --git a/providers/openrouter/models/openai/gpt-4o-mini-2024-07-18.toml b/providers/openrouter/models/openai/gpt-4o-mini-2024-07-18.toml new file mode 100644 index 000000000..6a61a3ffa --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-mini-2024-07-18.toml @@ -0,0 +1,24 @@ +name = "GPT-4o-mini (2024-07-18)" +family = "o-mini" +release_date = "2024-07-18" +last_updated = "2024-07-18" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.075 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-mini-search-preview.toml b/providers/openrouter/models/openai/gpt-4o-mini-search-preview.toml new file mode 100644 index 000000000..842dc5fcb --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-mini-search-preview.toml @@ -0,0 +1,23 @@ +name = "GPT-4o-mini Search Preview" +family = "o-mini" +release_date = "2025-03-12" +last_updated = "2025-03-12" +attachment = false +reasoning = false +temperature = false +tool_call = false +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 0.15 +output = 0.6 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-mini.toml b/providers/openrouter/models/openai/gpt-4o-mini.toml index 6b83dcf5b..c2a77a903 100644 --- a/providers/openrouter/models/openai/gpt-4o-mini.toml +++ b/providers/openrouter/models/openai/gpt-4o-mini.toml @@ -7,18 +7,18 @@ reasoning = false temperature = true tool_call = true structured_output = true -knowledge = "2024-10" +knowledge = "2023-10-31" open_weights = false [cost] input = 0.15 -output = 0.60 -cache_read = 0.08 +output = 0.6 +cache_read = 0.075 [limit] context = 128_000 output = 16_384 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o-search-preview.toml b/providers/openrouter/models/openai/gpt-4o-search-preview.toml new file mode 100644 index 000000000..9b1c0bb67 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o-search-preview.toml @@ -0,0 +1,23 @@ +name = "GPT-4o Search Preview" +family = "gpt" +release_date = "2025-03-12" +last_updated = "2025-03-12" +attachment = false +reasoning = false +temperature = false +tool_call = false +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-4o.toml b/providers/openrouter/models/openai/gpt-4o.toml new file mode 100644 index 000000000..d60e676a3 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-4o.toml @@ -0,0 +1,23 @@ +name = "GPT-4o" +family = "gpt" +release_date = "2024-05-13" +last_updated = "2024-05-13" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5-chat.toml b/providers/openrouter/models/openai/gpt-5-chat.toml index 9da5dcef6..d85e9fd34 100644 --- a/providers/openrouter/models/openai/gpt-5-chat.toml +++ b/providers/openrouter/models/openai/gpt-5-chat.toml @@ -1,23 +1,24 @@ -name = "GPT-5 Chat (latest)" +name = "GPT-5 Chat" family = "gpt-codex" release_date = "2025-08-07" last_updated = "2025-08-07" attachment = true -reasoning = true -temperature = true -knowledge = "2024-09-30" +reasoning = false +temperature = false tool_call = false structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 1.25 -output = 10.00 +output = 10 +cache_read = 0.125 [limit] -context = 400_000 -output = 128_000 +context = 128_000 +output = 16_384 [modalities] -input = ["text", "image"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5-codex.toml b/providers/openrouter/models/openai/gpt-5-codex.toml index 361135416..a0e7b9d20 100644 --- a/providers/openrouter/models/openai/gpt-5-codex.toml +++ b/providers/openrouter/models/openai/gpt-5-codex.toml @@ -1,18 +1,18 @@ name = "GPT-5 Codex" family = "gpt-codex" -release_date = "2025-09-15" -last_updated = "2025-09-15" +release_date = "2025-09-23" +last_updated = "2025-09-23" attachment = true reasoning = true -temperature = true -knowledge = "2024-10-01" +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 1.25 -output = 10.00 +output = 10 cache_read = 0.125 [limit] diff --git a/providers/openrouter/models/openai/gpt-5-image-mini.toml b/providers/openrouter/models/openai/gpt-5-image-mini.toml new file mode 100644 index 000000000..21569928d --- /dev/null +++ b/providers/openrouter/models/openai/gpt-5-image-mini.toml @@ -0,0 +1,23 @@ +name = "GPT-5 Image Mini" +family = "gpt" +release_date = "2025-10-16" +last_updated = "2025-10-16" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 2.5 +output = 2 +cache_read = 0.25 + +[limit] +context = 400_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["image", "text"] diff --git a/providers/openrouter/models/openai/gpt-5-image.toml b/providers/openrouter/models/openai/gpt-5-image.toml index ab3356cc4..44d7472ae 100644 --- a/providers/openrouter/models/openai/gpt-5-image.toml +++ b/providers/openrouter/models/openai/gpt-5-image.toml @@ -5,14 +5,14 @@ last_updated = "2025-10-14" attachment = true reasoning = true temperature = true -knowledge = "2024-10-01" -tool_call = true +tool_call = false structured_output = true +knowledge = "2024-10-01" open_weights = false [cost] -input = 5.00 -output = 10.00 +input = 10 +output = 10 cache_read = 1.25 [limit] @@ -20,5 +20,5 @@ context = 400_000 output = 128_000 [modalities] -input = ["text", "image", "pdf"] -output = ["text", "image"] \ No newline at end of file +input = ["image", "text", "pdf"] +output = ["image", "text"] diff --git a/providers/openrouter/models/openai/gpt-5-mini.toml b/providers/openrouter/models/openai/gpt-5-mini.toml index ef63fc159..39cb69c8b 100644 --- a/providers/openrouter/models/openai/gpt-5-mini.toml +++ b/providers/openrouter/models/openai/gpt-5-mini.toml @@ -4,20 +4,21 @@ release_date = "2025-08-07" last_updated = "2025-08-07" attachment = true reasoning = true -temperature = true -knowledge = "2024-10-01" +temperature = false tool_call = true structured_output = true +knowledge = "2024-05-31" open_weights = false [cost] input = 0.25 -output = 2.00 +output = 2 +cache_read = 0.025 [limit] context = 400_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5-nano.toml b/providers/openrouter/models/openai/gpt-5-nano.toml index 1bcd22091..22ca5b9e3 100644 --- a/providers/openrouter/models/openai/gpt-5-nano.toml +++ b/providers/openrouter/models/openai/gpt-5-nano.toml @@ -4,20 +4,21 @@ release_date = "2025-08-07" last_updated = "2025-08-07" attachment = true reasoning = true -temperature = true -knowledge = "2024-10-01" +temperature = false tool_call = true structured_output = true +knowledge = "2024-05-31" open_weights = false [cost] input = 0.05 -output = 0.40 +output = 0.4 +cache_read = 0.01 [limit] context = 400_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5-pro.toml b/providers/openrouter/models/openai/gpt-5-pro.toml index 3723e08eb..2c38d984e 100644 --- a/providers/openrouter/models/openai/gpt-5-pro.toml +++ b/providers/openrouter/models/openai/gpt-5-pro.toml @@ -5,19 +5,19 @@ last_updated = "2025-10-06" attachment = true reasoning = true temperature = false -knowledge = "2024-09-30" tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] -input = 15.00 -output = 120.00 +input = 15 +output = 120 [limit] context = 400_000 -output = 272_000 +output = 128_000 [modalities] -input = ["text", "image"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.1-chat.toml b/providers/openrouter/models/openai/gpt-5.1-chat.toml index 3dbce2282..80a89d644 100644 --- a/providers/openrouter/models/openai/gpt-5.1-chat.toml +++ b/providers/openrouter/models/openai/gpt-5.1-chat.toml @@ -3,16 +3,16 @@ family = "gpt-codex" release_date = "2025-11-13" last_updated = "2025-11-13" attachment = true -reasoning = true -temperature = true -knowledge = "2024-09-30" +reasoning = false +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 1.25 -output = 10.00 +output = 10 cache_read = 0.125 [limit] @@ -20,5 +20,5 @@ context = 128_000 output = 16_384 [modalities] -input = ["text", "image"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.1-codex-max.toml b/providers/openrouter/models/openai/gpt-5.1-codex-max.toml index 5e160b6f0..5c171a890 100644 --- a/providers/openrouter/models/openai/gpt-5.1-codex-max.toml +++ b/providers/openrouter/models/openai/gpt-5.1-codex-max.toml @@ -1,19 +1,19 @@ name = "GPT-5.1-Codex-Max" family = "gpt-codex" -release_date = "2025-11-13" -last_updated = "2025-11-13" +release_date = "2025-12-04" +last_updated = "2025-12-04" attachment = true reasoning = true -temperature = true -knowledge = "2024-09-30" +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] -input = 1.10 -output = 9.00 -cache_read = 0.11 +input = 1.25 +output = 10 +cache_read = 0.125 [limit] context = 400_000 diff --git a/providers/openrouter/models/openai/gpt-5.1-codex-mini.toml b/providers/openrouter/models/openai/gpt-5.1-codex-mini.toml index 6a44be52c..61cf7b474 100644 --- a/providers/openrouter/models/openai/gpt-5.1-codex-mini.toml +++ b/providers/openrouter/models/openai/gpt-5.1-codex-mini.toml @@ -4,21 +4,21 @@ release_date = "2025-11-13" last_updated = "2025-11-13" attachment = true reasoning = true -temperature = true -knowledge = "2024-09-30" +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 0.25 -output = 2.00 -cache_read = 0.025 +output = 2 +cache_read = 0.03 [limit] context = 400_000 -output = 100_000 +output = 128_000 [modalities] -input = ["text", "image"] +input = ["image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.1-codex.toml b/providers/openrouter/models/openai/gpt-5.1-codex.toml index 110b7c0e5..081843db6 100644 --- a/providers/openrouter/models/openai/gpt-5.1-codex.toml +++ b/providers/openrouter/models/openai/gpt-5.1-codex.toml @@ -4,15 +4,15 @@ release_date = "2025-11-13" last_updated = "2025-11-13" attachment = true reasoning = true -temperature = true -knowledge = "2024-09-30" +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 1.25 -output = 10.00 +output = 10 cache_read = 0.125 [limit] diff --git a/providers/openrouter/models/openai/gpt-5.1.toml b/providers/openrouter/models/openai/gpt-5.1.toml index e617a4513..73a19762d 100644 --- a/providers/openrouter/models/openai/gpt-5.1.toml +++ b/providers/openrouter/models/openai/gpt-5.1.toml @@ -4,21 +4,21 @@ release_date = "2025-11-13" last_updated = "2025-11-13" attachment = true reasoning = true -temperature = true -knowledge = "2024-09-30" +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 1.25 -output = 10.00 -cache_read = 0.125 +output = 10 +cache_read = 0.13 [limit] context = 400_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.2-chat.toml b/providers/openrouter/models/openai/gpt-5.2-chat.toml index 6d128990b..66ccd9316 100644 --- a/providers/openrouter/models/openai/gpt-5.2-chat.toml +++ b/providers/openrouter/models/openai/gpt-5.2-chat.toml @@ -1,24 +1,24 @@ name = "GPT-5.2 Chat" family = "gpt-codex" -release_date = "2025-12-11" -last_updated = "2025-12-11" +release_date = "2025-12-10" +last_updated = "2025-12-10" attachment = true -reasoning = true +reasoning = false temperature = false -knowledge = "2025-08-31" tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 [limit] context = 128_000 -output = 16_384 +output = 32_000 [modalities] -input = ["text", "image"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.2-codex.toml b/providers/openrouter/models/openai/gpt-5.2-codex.toml index 28b76fa4d..76c089c17 100644 --- a/providers/openrouter/models/openai/gpt-5.2-codex.toml +++ b/providers/openrouter/models/openai/gpt-5.2-codex.toml @@ -4,15 +4,15 @@ release_date = "2026-01-14" last_updated = "2026-01-14" attachment = true reasoning = true -temperature = true -knowledge = "2025-08-31" +temperature = false tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 [limit] diff --git a/providers/openrouter/models/openai/gpt-5.2-pro.toml b/providers/openrouter/models/openai/gpt-5.2-pro.toml index 83b05e96f..20339eeb2 100644 --- a/providers/openrouter/models/openai/gpt-5.2-pro.toml +++ b/providers/openrouter/models/openai/gpt-5.2-pro.toml @@ -1,23 +1,23 @@ name = "GPT-5.2 Pro" family = "gpt-pro" -release_date = "2025-12-11" -last_updated = "2025-12-11" +release_date = "2025-12-10" +last_updated = "2025-12-10" attachment = true reasoning = true temperature = false -knowledge = "2025-08-31" tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] -input = 21.00 -output = 168.00 +input = 21 +output = 168 [limit] context = 400_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.2.toml b/providers/openrouter/models/openai/gpt-5.2.toml index ff0612d3d..0b17f1f62 100644 --- a/providers/openrouter/models/openai/gpt-5.2.toml +++ b/providers/openrouter/models/openai/gpt-5.2.toml @@ -1,18 +1,18 @@ name = "GPT-5.2" family = "gpt" -release_date = "2025-12-11" -last_updated = "2025-12-11" +release_date = "2025-12-10" +last_updated = "2025-12-10" attachment = true reasoning = true temperature = false -knowledge = "2025-08-31" tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 [limit] @@ -20,5 +20,5 @@ context = 400_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.3-chat.toml b/providers/openrouter/models/openai/gpt-5.3-chat.toml new file mode 100644 index 000000000..f39874205 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-5.3-chat.toml @@ -0,0 +1,23 @@ +name = "GPT-5.3 Chat" +family = "gpt" +release_date = "2026-03-03" +last_updated = "2026-03-03" +attachment = true +reasoning = false +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.3-codex.toml b/providers/openrouter/models/openai/gpt-5.3-codex.toml index 3004518b8..257c52359 100644 --- a/providers/openrouter/models/openai/gpt-5.3-codex.toml +++ b/providers/openrouter/models/openai/gpt-5.3-codex.toml @@ -5,14 +5,14 @@ last_updated = "2026-02-24" attachment = true reasoning = true temperature = false -knowledge = "2025-08-31" tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 [limit] diff --git a/providers/openrouter/models/openai/gpt-5.4-image-2.toml b/providers/openrouter/models/openai/gpt-5.4-image-2.toml new file mode 100644 index 000000000..27bb49639 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-5.4-image-2.toml @@ -0,0 +1,23 @@ +name = "GPT-5.4 Image 2" +family = "gpt" +release_date = "2026-04-21" +last_updated = "2026-04-21" +attachment = true +reasoning = true +temperature = false +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 8 +output = 15 +cache_read = 2 + +[limit] +context = 272_000 +output = 128_000 + +[modalities] +input = ["image", "text", "pdf"] +output = ["image", "text"] diff --git a/providers/openrouter/models/openai/gpt-5.4-mini.toml b/providers/openrouter/models/openai/gpt-5.4-mini.toml index 5788c53a2..1ffc18851 100644 --- a/providers/openrouter/models/openai/gpt-5.4-mini.toml +++ b/providers/openrouter/models/openai/gpt-5.4-mini.toml @@ -4,10 +4,10 @@ release_date = "2026-03-17" last_updated = "2026-03-17" attachment = true reasoning = true -temperature = true -knowledge = "2025-08-31" +temperature = false tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] @@ -20,5 +20,5 @@ context = 400_000 output = 128_000 [modalities] -input = ["text", "image", "pdf"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.4-nano.toml b/providers/openrouter/models/openai/gpt-5.4-nano.toml index bd7b52de6..bde72c3f6 100644 --- a/providers/openrouter/models/openai/gpt-5.4-nano.toml +++ b/providers/openrouter/models/openai/gpt-5.4-nano.toml @@ -3,15 +3,15 @@ family = "gpt-nano" release_date = "2026-03-17" last_updated = "2026-03-17" attachment = true -reasoning = false -temperature = true -knowledge = "2025-08-31" +reasoning = true +temperature = false tool_call = true structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] -input = 0.20 +input = 0.2 output = 1.25 cache_read = 0.02 @@ -20,5 +20,5 @@ context = 400_000 output = 128_000 [modalities] -input = ["text", "image", "pdf"] +input = ["pdf", "image", "text"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.4-pro.toml b/providers/openrouter/models/openai/gpt-5.4-pro.toml index e1b44ec2b..70bfcdcfc 100644 --- a/providers/openrouter/models/openai/gpt-5.4-pro.toml +++ b/providers/openrouter/models/openai/gpt-5.4-pro.toml @@ -5,15 +5,14 @@ last_updated = "2026-03-05" attachment = true reasoning = true temperature = false -knowledge = "2025-08-31" tool_call = true -structured_output = false +structured_output = true +knowledge = "2025-08-31" open_weights = false [cost] -input = 30.00 -output = 180.00 -cache_read = 30.00 +input = 30 +output = 180 [limit] context = 1_050_000 diff --git a/providers/openrouter/models/openai/gpt-5.4.toml b/providers/openrouter/models/openai/gpt-5.4.toml index 706aa2673..2a74a99b5 100644 --- a/providers/openrouter/models/openai/gpt-5.4.toml +++ b/providers/openrouter/models/openai/gpt-5.4.toml @@ -1,3 +1,23 @@ -[extends] -from = "openai/gpt-5.4" -omit = ["experimental.modes.fast"] +name = "GPT-5.4" +family = "gpt" +release_date = "2026-03-05" +last_updated = "2026-03-05" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 2.5 +output = 15 +cache_read = 0.25 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.5-pro.toml b/providers/openrouter/models/openai/gpt-5.5-pro.toml index 97bc59ce0..52bfd1749 100644 --- a/providers/openrouter/models/openai/gpt-5.5-pro.toml +++ b/providers/openrouter/models/openai/gpt-5.5-pro.toml @@ -1,2 +1,23 @@ -[extends] -from = "openai/gpt-5.5-pro" +name = "GPT-5.5 Pro" +family = "gpt" +release_date = "2026-04-24" +last_updated = "2026-04-24" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2025-12-01" +open_weights = false + +[cost] +input = 30 +output = 180 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.5.toml b/providers/openrouter/models/openai/gpt-5.5.toml index 60eceb01a..9178db06f 100644 --- a/providers/openrouter/models/openai/gpt-5.5.toml +++ b/providers/openrouter/models/openai/gpt-5.5.toml @@ -1,3 +1,24 @@ -[extends] -from = "openai/gpt-5.5" -omit = ["experimental.modes.fast"] +name = "GPT-5.5" +family = "gpt" +release_date = "2026-04-24" +last_updated = "2026-04-24" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2025-12-01" +open_weights = false + +[cost] +input = 5 +output = 30 +cache_read = 0.5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-5.toml b/providers/openrouter/models/openai/gpt-5.toml index 5cfc8027b..fee43091c 100644 --- a/providers/openrouter/models/openai/gpt-5.toml +++ b/providers/openrouter/models/openai/gpt-5.toml @@ -4,20 +4,21 @@ release_date = "2025-08-07" last_updated = "2025-08-07" attachment = true reasoning = true -temperature = true -knowledge = "2024-10-01" +temperature = false tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false [cost] input = 1.25 -output = 10.00 +output = 10 +cache_read = 0.125 [limit] context = 400_000 output = 128_000 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-audio-mini.toml b/providers/openrouter/models/openai/gpt-audio-mini.toml new file mode 100644 index 000000000..8b70d3601 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-audio-mini.toml @@ -0,0 +1,22 @@ +name = "GPT Audio Mini" +family = "o-mini" +release_date = "2026-01-19" +last_updated = "2026-01-19" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.6 +output = 2.4 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "audio"] +output = ["text", "audio"] diff --git a/providers/openrouter/models/openai/gpt-audio.toml b/providers/openrouter/models/openai/gpt-audio.toml new file mode 100644 index 000000000..1f8d186e3 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-audio.toml @@ -0,0 +1,22 @@ +name = "GPT Audio" +family = "gpt" +release_date = "2026-01-19" +last_updated = "2026-01-19" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 2.5 +output = 10 + +[limit] +context = 128_000 +output = 16_384 + +[modalities] +input = ["text", "audio"] +output = ["text", "audio"] diff --git a/providers/openrouter/models/openai/gpt-chat-latest.toml b/providers/openrouter/models/openai/gpt-chat-latest.toml new file mode 100644 index 000000000..a920a36a2 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-chat-latest.toml @@ -0,0 +1,23 @@ +name = "GPT Chat Latest" +family = "gpt" +release_date = "2026-05-05" +last_updated = "2026-05-05" +attachment = true +reasoning = false +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 5 +output = 30 +cache_read = 0.5 + +[limit] +context = 400_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-oss-120b.toml b/providers/openrouter/models/openai/gpt-oss-120b.toml index 62e3695cd..3dbb090f4 100644 --- a/providers/openrouter/models/openai/gpt-oss-120b.toml +++ b/providers/openrouter/models/openai/gpt-oss-120b.toml @@ -1,4 +1,4 @@ -name = "GPT OSS 120B" +name = "gpt-oss-120b" family = "gpt-oss" release_date = "2025-08-05" last_updated = "2025-08-05" @@ -7,11 +7,12 @@ reasoning = true temperature = true tool_call = true structured_output = true +knowledge = "2024-06-30" open_weights = true [cost] -input = 0.072 -output = 0.28 +input = 0.039 +output = 0.18 [limit] context = 131_072 diff --git a/providers/openrouter/models/openai/gpt-oss-120b:free.toml b/providers/openrouter/models/openai/gpt-oss-120b:free.toml index 3030f1635..771ed073e 100644 --- a/providers/openrouter/models/openai/gpt-oss-120b:free.toml +++ b/providers/openrouter/models/openai/gpt-oss-120b:free.toml @@ -6,15 +6,17 @@ attachment = false reasoning = true temperature = true tool_call = true +structured_output = false +knowledge = "2024-06-30" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 131_072 -output = 32_768 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/openai/gpt-oss-20b.toml b/providers/openrouter/models/openai/gpt-oss-20b.toml index 916314f7b..9b243698a 100644 --- a/providers/openrouter/models/openai/gpt-oss-20b.toml +++ b/providers/openrouter/models/openai/gpt-oss-20b.toml @@ -1,4 +1,4 @@ -name = "GPT OSS 20B" +name = "gpt-oss-20b" family = "gpt-oss" release_date = "2025-08-05" last_updated = "2025-08-05" @@ -7,15 +7,16 @@ reasoning = true temperature = true tool_call = true structured_output = true +knowledge = "2024-06-30" open_weights = true [cost] -input = 0.05 -output = 0.20 +input = 0.03 +output = 0.14 [limit] context = 131_072 -output = 32_768 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/openai/gpt-oss-20b:free.toml b/providers/openrouter/models/openai/gpt-oss-20b:free.toml index ff92493d0..1d71d485d 100644 --- a/providers/openrouter/models/openai/gpt-oss-20b:free.toml +++ b/providers/openrouter/models/openai/gpt-oss-20b:free.toml @@ -1,20 +1,22 @@ name = "gpt-oss-20b (free)" family = "gpt-oss" release_date = "2025-08-05" -last_updated = "2026-01-31" +last_updated = "2025-08-05" attachment = false reasoning = true temperature = true tool_call = true +structured_output = false +knowledge = "2024-06-30" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 131_072 -output = 32_768 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/openrouter/models/openai/gpt-oss-safeguard-20b.toml b/providers/openrouter/models/openai/gpt-oss-safeguard-20b.toml index 5386514cc..3f21d44d3 100644 --- a/providers/openrouter/models/openai/gpt-oss-safeguard-20b.toml +++ b/providers/openrouter/models/openai/gpt-oss-safeguard-20b.toml @@ -1,4 +1,4 @@ -name = "GPT OSS Safeguard 20B" +name = "gpt-oss-safeguard-20b" family = "gpt-oss" release_date = "2025-10-29" last_updated = "2025-10-29" @@ -6,11 +6,13 @@ attachment = false reasoning = true temperature = true tool_call = true -open_weights = false +structured_output = true +open_weights = true [cost] input = 0.075 -output = 0.30 +output = 0.3 +cache_read = 0.037 [limit] context = 131_072 diff --git a/providers/openrouter/models/openai/o1-pro.toml b/providers/openrouter/models/openai/o1-pro.toml new file mode 100644 index 000000000..3291554a8 --- /dev/null +++ b/providers/openrouter/models/openai/o1-pro.toml @@ -0,0 +1,23 @@ +name = "o1-pro" +family = "o" +release_date = "2025-03-19" +last_updated = "2025-03-19" +attachment = true +reasoning = true +temperature = false +tool_call = false +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 150 +output = 600 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o1.toml b/providers/openrouter/models/openai/o1.toml new file mode 100644 index 000000000..dade0701d --- /dev/null +++ b/providers/openrouter/models/openai/o1.toml @@ -0,0 +1,24 @@ +name = "o1" +family = "o" +release_date = "2024-12-17" +last_updated = "2024-12-17" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 15 +output = 60 +cache_read = 7.5 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o3-deep-research.toml b/providers/openrouter/models/openai/o3-deep-research.toml new file mode 100644 index 000000000..6f61d9d1e --- /dev/null +++ b/providers/openrouter/models/openai/o3-deep-research.toml @@ -0,0 +1,23 @@ +name = "o3 Deep Research" +family = "o" +release_date = "2025-10-10" +last_updated = "2025-10-10" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 10 +output = 40 +cache_read = 2.5 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["image", "text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o3-mini-high.toml b/providers/openrouter/models/openai/o3-mini-high.toml new file mode 100644 index 000000000..c856538b6 --- /dev/null +++ b/providers/openrouter/models/openai/o3-mini-high.toml @@ -0,0 +1,24 @@ +name = "o3 Mini High" +family = "o" +release_date = "2025-02-12" +last_updated = "2025-02-12" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.55 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o3-mini.toml b/providers/openrouter/models/openai/o3-mini.toml new file mode 100644 index 000000000..1900c314e --- /dev/null +++ b/providers/openrouter/models/openai/o3-mini.toml @@ -0,0 +1,24 @@ +name = "o3 Mini" +family = "o" +release_date = "2025-01-31" +last_updated = "2025-01-31" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2023-10-31" +open_weights = false + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.55 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o3-pro.toml b/providers/openrouter/models/openai/o3-pro.toml new file mode 100644 index 000000000..8d1b711a4 --- /dev/null +++ b/providers/openrouter/models/openai/o3-pro.toml @@ -0,0 +1,23 @@ +name = "o3 Pro" +family = "o" +release_date = "2025-06-10" +last_updated = "2025-06-10" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2024-06-30" +open_weights = false + +[cost] +input = 20 +output = 80 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["text", "pdf", "image"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o3.toml b/providers/openrouter/models/openai/o3.toml new file mode 100644 index 000000000..3e4e1af23 --- /dev/null +++ b/providers/openrouter/models/openai/o3.toml @@ -0,0 +1,24 @@ +name = "o3" +family = "o" +release_date = "2025-04-16" +last_updated = "2025-04-16" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2024-06-30" +open_weights = false + +[cost] +input = 2 +output = 8 +cache_read = 0.5 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["image", "text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o4-mini-deep-research.toml b/providers/openrouter/models/openai/o4-mini-deep-research.toml new file mode 100644 index 000000000..0467e1654 --- /dev/null +++ b/providers/openrouter/models/openai/o4-mini-deep-research.toml @@ -0,0 +1,23 @@ +name = "o4 Mini Deep Research" +family = "o" +release_date = "2025-10-10" +last_updated = "2025-10-10" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 2 +output = 8 +cache_read = 0.5 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o4-mini-high.toml b/providers/openrouter/models/openai/o4-mini-high.toml new file mode 100644 index 000000000..8a39d460a --- /dev/null +++ b/providers/openrouter/models/openai/o4-mini-high.toml @@ -0,0 +1,24 @@ +name = "o4 Mini High" +family = "o" +release_date = "2025-04-16" +last_updated = "2025-04-16" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2024-06-30" +open_weights = false + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.275 + +[limit] +context = 200_000 +output = 100_000 + +[modalities] +input = ["image", "text", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/openai/o4-mini.toml b/providers/openrouter/models/openai/o4-mini.toml index b13e369c4..dcdda5dc7 100644 --- a/providers/openrouter/models/openai/o4-mini.toml +++ b/providers/openrouter/models/openai/o4-mini.toml @@ -4,21 +4,21 @@ release_date = "2025-04-16" last_updated = "2025-04-16" attachment = true reasoning = true -temperature = true +temperature = false tool_call = true structured_output = true -knowledge = "2024-06" +knowledge = "2024-06-30" open_weights = false [cost] -input = 1.10 -output = 4.40 -cache_read = 0.28 +input = 1.1 +output = 4.4 +cache_read = 0.275 [limit] context = 200_000 output = 100_000 [modalities] -input = ["text", "image"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/openrouter/auto.toml b/providers/openrouter/models/openrouter/auto.toml new file mode 100644 index 000000000..62bdd14e4 --- /dev/null +++ b/providers/openrouter/models/openrouter/auto.toml @@ -0,0 +1,18 @@ +name = "Auto Router" +family = "auto" +release_date = "2023-11-08" +last_updated = "2023-11-08" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[limit] +context = 2_000_000 +output = 2_000_000 + +[modalities] +input = ["text", "image", "audio", "pdf", "video"] +output = ["text", "image"] diff --git a/providers/openrouter/models/openrouter/bodybuilder.toml b/providers/openrouter/models/openrouter/bodybuilder.toml new file mode 100644 index 000000000..71b63e43a --- /dev/null +++ b/providers/openrouter/models/openrouter/bodybuilder.toml @@ -0,0 +1,18 @@ +name = "Body Builder (beta)" +family = "o" +release_date = "2025-12-05" +last_updated = "2025-12-05" +attachment = false +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/openrouter/free.toml b/providers/openrouter/models/openrouter/free.toml index 885b28ccb..01c98dc26 100644 --- a/providers/openrouter/models/openrouter/free.toml +++ b/providers/openrouter/models/openrouter/free.toml @@ -1,4 +1,5 @@ name = "Free Models Router" +family = "o" release_date = "2026-02-01" last_updated = "2026-02-01" attachment = true @@ -9,8 +10,8 @@ structured_output = true open_weights = false [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] context = 200_000 diff --git a/providers/openrouter/models/openrouter/owl-alpha.toml b/providers/openrouter/models/openrouter/owl-alpha.toml index 21a33fbd3..982668039 100644 --- a/providers/openrouter/models/openrouter/owl-alpha.toml +++ b/providers/openrouter/models/openrouter/owl-alpha.toml @@ -1,8 +1,9 @@ name = "Owl Alpha" +family = "alpha" release_date = "2026-04-28" -last_updated = "2026-04-30" +last_updated = "2026-04-28" attachment = false -reasoning = true +reasoning = false temperature = true tool_call = true structured_output = true @@ -14,8 +15,8 @@ input = 0 output = 0 [limit] -context = 1048756 -output = 262144 +context = 1_048_756 +output = 262_144 [modalities] input = ["text"] diff --git a/providers/openrouter/models/openrouter/pareto-code.toml b/providers/openrouter/models/openrouter/pareto-code.toml index 0c1565366..2eedc0c0f 100644 --- a/providers/openrouter/models/openrouter/pareto-code.toml +++ b/providers/openrouter/models/openrouter/pareto-code.toml @@ -1,15 +1,16 @@ name = "Pareto Code Router" +family = "o" release_date = "2026-04-21" last_updated = "2026-04-21" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = true +attachment = false +reasoning = false +temperature = false +tool_call = false +structured_output = false open_weights = false [limit] -context = 200_000 +context = 2_000_000 output = 200_000 [modalities] diff --git a/providers/openrouter/models/perceptron/perceptron-mk1.toml b/providers/openrouter/models/perceptron/perceptron-mk1.toml new file mode 100644 index 000000000..34e5648c0 --- /dev/null +++ b/providers/openrouter/models/perceptron/perceptron-mk1.toml @@ -0,0 +1,22 @@ +name = "Perceptron Mk1" +family = "o" +release_date = "2026-05-12" +last_updated = "2026-05-12" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 0.15 +output = 1.5 + +[limit] +context = 32_768 +output = 8_192 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/perplexity/sonar-deep-research.toml b/providers/openrouter/models/perplexity/sonar-deep-research.toml new file mode 100644 index 000000000..b785098e6 --- /dev/null +++ b/providers/openrouter/models/perplexity/sonar-deep-research.toml @@ -0,0 +1,23 @@ +name = "Sonar Deep Research" +family = "sonar-deep-research" +release_date = "2025-03-07" +last_updated = "2025-03-07" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 2 +output = 8 +reasoning = 3 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/perplexity/sonar-pro-search.toml b/providers/openrouter/models/perplexity/sonar-pro-search.toml new file mode 100644 index 000000000..f58bf3e26 --- /dev/null +++ b/providers/openrouter/models/perplexity/sonar-pro-search.toml @@ -0,0 +1,22 @@ +name = "Sonar Pro Search" +family = "sonar-pro" +release_date = "2025-10-30" +last_updated = "2025-10-30" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[cost] +input = 3 +output = 15 + +[limit] +context = 200_000 +output = 8_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/perplexity/sonar-pro.toml b/providers/openrouter/models/perplexity/sonar-pro.toml new file mode 100644 index 000000000..d0596d1da --- /dev/null +++ b/providers/openrouter/models/perplexity/sonar-pro.toml @@ -0,0 +1,22 @@ +name = "Sonar Pro" +family = "sonar-pro" +release_date = "2025-03-07" +last_updated = "2025-03-07" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 3 +output = 15 + +[limit] +context = 200_000 +output = 8_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/perplexity/sonar-reasoning-pro.toml b/providers/openrouter/models/perplexity/sonar-reasoning-pro.toml new file mode 100644 index 000000000..9c47a0276 --- /dev/null +++ b/providers/openrouter/models/perplexity/sonar-reasoning-pro.toml @@ -0,0 +1,22 @@ +name = "Sonar Reasoning Pro" +family = "sonar-reasoning" +release_date = "2025-03-07" +last_updated = "2025-03-07" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 2 +output = 8 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/perplexity/sonar.toml b/providers/openrouter/models/perplexity/sonar.toml new file mode 100644 index 000000000..f4bcb58e2 --- /dev/null +++ b/providers/openrouter/models/perplexity/sonar.toml @@ -0,0 +1,22 @@ +name = "Sonar" +family = "sonar" +release_date = "2025-01-27" +last_updated = "2025-01-27" +attachment = true +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 1 +output = 1 + +[limit] +context = 127_072 +output = 127_072 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/poolside/laguna-m.1:free.toml b/providers/openrouter/models/poolside/laguna-m.1:free.toml index 4ce492038..c7e6f5d8c 100644 --- a/providers/openrouter/models/poolside/laguna-m.1:free.toml +++ b/providers/openrouter/models/poolside/laguna-m.1:free.toml @@ -1,20 +1,20 @@ -name = "Laguna M.1" +name = "Laguna M.1 (free)" +family = "o" release_date = "2026-04-28" last_updated = "2026-04-28" attachment = false reasoning = true temperature = true tool_call = true +structured_output = false open_weights = false [interleaved] field = "reasoning_content" [cost] -input = 0.0 -output = 0.0 -cache_read = 0.0 -cache_write = 0.0 +input = 0 +output = 0 [limit] context = 131_072 @@ -22,4 +22,4 @@ output = 8_192 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/poolside/laguna-xs.2:free.toml b/providers/openrouter/models/poolside/laguna-xs.2:free.toml index fcebe340e..0a47cde17 100644 --- a/providers/openrouter/models/poolside/laguna-xs.2:free.toml +++ b/providers/openrouter/models/poolside/laguna-xs.2:free.toml @@ -1,20 +1,20 @@ -name = "Laguna XS.2" +name = "Laguna XS.2 (free)" +family = "o" release_date = "2026-04-28" last_updated = "2026-04-28" attachment = false reasoning = true temperature = true tool_call = true +structured_output = false open_weights = true [interleaved] field = "reasoning_content" [cost] -input = 0.0 -output = 0.0 -cache_read = 0.0 -cache_write = 0.0 +input = 0 +output = 0 [limit] context = 131_072 @@ -22,4 +22,4 @@ output = 8_192 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/prime-intellect/intellect-3.toml b/providers/openrouter/models/prime-intellect/intellect-3.toml index 13cb7c8cc..c8555bc7c 100644 --- a/providers/openrouter/models/prime-intellect/intellect-3.toml +++ b/providers/openrouter/models/prime-intellect/intellect-3.toml @@ -1,17 +1,22 @@ -id = "prime-intellect/intellect-3" -name = "Intellect 3" +name = "INTELLECT-3" family = "glm" -release_date = "2025-01-15" -last_updated = "2025-01-15" +release_date = "2025-11-27" +last_updated = "2025-11-27" attachment = false reasoning = true temperature = true -knowledge = "2024-10" tool_call = true structured_output = true +knowledge = "2024-10" open_weights = true -cost = { input = 0.2, output = 1.1 } -limit = { context = 131072, output = 8192 } + +[cost] +input = 0.2 +output = 1.1 + +[limit] +context = 131_072 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-2.5-72b-instruct.toml b/providers/openrouter/models/qwen/qwen-2.5-72b-instruct.toml new file mode 100644 index 000000000..58ff8d79d --- /dev/null +++ b/providers/openrouter/models/qwen/qwen-2.5-72b-instruct.toml @@ -0,0 +1,23 @@ +name = "Qwen2.5 72B Instruct" +family = "qwen" +release_date = "2024-09-19" +last_updated = "2024-09-19" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-06-30" +open_weights = true + +[cost] +input = 0.36 +output = 0.4 + +[limit] +context = 32_768 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-2.5-7b-instruct.toml b/providers/openrouter/models/qwen/qwen-2.5-7b-instruct.toml new file mode 100644 index 000000000..f5b6485b8 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen-2.5-7b-instruct.toml @@ -0,0 +1,23 @@ +name = "Qwen2.5 7B Instruct" +family = "qwen" +release_date = "2024-10-16" +last_updated = "2024-10-16" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-06-30" +open_weights = true + +[cost] +input = 0.04 +output = 0.1 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-2.5-coder-32b-instruct.toml b/providers/openrouter/models/qwen/qwen-2.5-coder-32b-instruct.toml index 534b89ef5..83bac4f02 100644 --- a/providers/openrouter/models/qwen/qwen-2.5-coder-32b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen-2.5-coder-32b-instruct.toml @@ -1,4 +1,3 @@ -id = "qwen/qwen-2.5-coder-32b-instruct:free" name = "Qwen2.5 Coder 32B Instruct" family = "qwen" release_date = "2024-11-11" @@ -6,13 +5,19 @@ last_updated = "2024-11-11" attachment = false reasoning = false temperature = true -knowledge = "2024-10" tool_call = false -structured_output = true +structured_output = false +knowledge = "2024-06-30" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 32768, output = 8192 } + +[cost] +input = 0.66 +output = 1 + +[limit] +context = 32_768 +output = 8_192 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-plus-2025-07-28.toml b/providers/openrouter/models/qwen/qwen-plus-2025-07-28.toml new file mode 100644 index 000000000..77f5fb73c --- /dev/null +++ b/providers/openrouter/models/qwen/qwen-plus-2025-07-28.toml @@ -0,0 +1,24 @@ +name = "Qwen Plus 0728" +family = "qwen" +release_date = "2025-09-08" +last_updated = "2025-09-08" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = false + +[cost] +input = 0.26 +output = 0.78 +cache_write = 0.325 + +[limit] +context = 1_000_000 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-plus-2025-07-28:thinking.toml b/providers/openrouter/models/qwen/qwen-plus-2025-07-28:thinking.toml new file mode 100644 index 000000000..ba9a5c4de --- /dev/null +++ b/providers/openrouter/models/qwen/qwen-plus-2025-07-28:thinking.toml @@ -0,0 +1,24 @@ +name = "Qwen Plus 0728 (thinking)" +family = "qwen" +release_date = "2025-09-08" +last_updated = "2025-09-08" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = false + +[cost] +input = 0.26 +output = 0.78 +cache_write = 0.325 + +[limit] +context = 1_000_000 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-plus.toml b/providers/openrouter/models/qwen/qwen-plus.toml index be2786647..a6df0814e 100644 --- a/providers/openrouter/models/qwen/qwen-plus.toml +++ b/providers/openrouter/models/qwen/qwen-plus.toml @@ -1,13 +1,13 @@ -name = "Qwen: Qwen-Plus" +name = "Qwen-Plus" family = "qwen" -release_date = "2025-01-25" -last_updated = "2025-01-25" +release_date = "2025-02-01" +last_updated = "2025-02-01" attachment = false reasoning = false temperature = true -knowledge = "2024-04" tool_call = true -structured_output = false +structured_output = true +knowledge = "2025-03-31" open_weights = false [cost] diff --git a/providers/openrouter/models/qwen/qwen2.5-vl-72b-instruct.toml b/providers/openrouter/models/qwen/qwen2.5-vl-72b-instruct.toml index 8b306c767..9d8994ba4 100644 --- a/providers/openrouter/models/qwen/qwen2.5-vl-72b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen2.5-vl-72b-instruct.toml @@ -1,4 +1,3 @@ -id = "qwen/qwen2.5-vl-72b-instruct:free" name = "Qwen2.5 VL 72B Instruct" family = "qwen" release_date = "2025-02-01" @@ -6,13 +5,19 @@ last_updated = "2025-02-01" attachment = true reasoning = false temperature = true -knowledge = "2024-10" tool_call = false structured_output = true +knowledge = "2024-06-30" open_weights = true -cost = { input = 0, output = 0 } -limit = { context = 32768, output = 8192 } + +[cost] +input = 0.25 +output = 0.75 + +[limit] +context = 32_000 +output = 8_192 [modalities] input = ["text", "image"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-14b.toml b/providers/openrouter/models/qwen/qwen3-14b.toml new file mode 100644 index 000000000..4da14ee76 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-14b.toml @@ -0,0 +1,23 @@ +name = "Qwen3 14B" +family = "qwen" +release_date = "2025-04-28" +last_updated = "2025-04-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.1 +output = 0.24 + +[limit] +context = 40_960 +output = 40_960 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-235b-a22b-07-25.toml b/providers/openrouter/models/qwen/qwen3-235b-a22b-2507.toml similarity index 70% rename from providers/openrouter/models/qwen/qwen3-235b-a22b-07-25.toml rename to providers/openrouter/models/qwen/qwen3-235b-a22b-2507.toml index d4f537ce6..94e056972 100644 --- a/providers/openrouter/models/qwen/qwen3-235b-a22b-07-25.toml +++ b/providers/openrouter/models/qwen/qwen3-235b-a22b-2507.toml @@ -1,23 +1,23 @@ name = "Qwen3 235B A22B Instruct 2507" family = "qwen" -release_date = "2025-04-28" +release_date = "2025-07-21" last_updated = "2025-07-21" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = true [cost] -input = 0.15 -output = 0.85 +input = 0.071 +output = 0.1 [limit] context = 262_144 -output = 131_072 +output = 16_384 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-235b-a22b-thinking-2507.toml b/providers/openrouter/models/qwen/qwen3-235b-a22b-thinking-2507.toml index be0ce5407..a8e511a8b 100644 --- a/providers/openrouter/models/qwen/qwen3-235b-a22b-thinking-2507.toml +++ b/providers/openrouter/models/qwen/qwen3-235b-a22b-thinking-2507.toml @@ -5,17 +5,17 @@ last_updated = "2025-07-25" attachment = false reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = true [cost] -input = 0.078 -output = 0.312 +input = 0.1495 +output = 1.495 [limit] -context = 262_144 +context = 131_072 output = 81_920 [modalities] diff --git a/providers/openrouter/models/qwen/qwen3-235b-a22b.toml b/providers/openrouter/models/qwen/qwen3-235b-a22b.toml new file mode 100644 index 000000000..05d18d6af --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-235b-a22b.toml @@ -0,0 +1,23 @@ +name = "Qwen3 235B A22B" +family = "qwen" +release_date = "2025-04-28" +last_updated = "2025-04-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.455 +output = 1.82 + +[limit] +context = 131_072 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-30b-a3b-instruct-2507.toml b/providers/openrouter/models/qwen/qwen3-30b-a3b-instruct-2507.toml index 24de613e5..ba98cecf3 100644 --- a/providers/openrouter/models/qwen/qwen3-30b-a3b-instruct-2507.toml +++ b/providers/openrouter/models/qwen/qwen3-30b-a3b-instruct-2507.toml @@ -5,18 +5,18 @@ last_updated = "2025-07-29" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = true [cost] -input = 0.20 -output = 0.80 +input = 0.09 +output = 0.3 [limit] -context = 262_000 -output = 262_000 +context = 262_144 +output = 262_144 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-30b-a3b-thinking-2507.toml b/providers/openrouter/models/qwen/qwen3-30b-a3b-thinking-2507.toml index c304950c7..72a1fd93e 100644 --- a/providers/openrouter/models/qwen/qwen3-30b-a3b-thinking-2507.toml +++ b/providers/openrouter/models/qwen/qwen3-30b-a3b-thinking-2507.toml @@ -1,22 +1,23 @@ name = "Qwen3 30B A3B Thinking 2507" family = "qwen" -release_date = "2025-07-29" -last_updated = "2025-07-29" +release_date = "2025-08-28" +last_updated = "2025-08-28" attachment = false reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = true [cost] -input = 0.20 -output = 0.80 +input = 0.08 +output = 0.4 +cache_read = 0.08 [limit] -context = 262_000 -output = 262_000 +context = 131_072 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-30b-a3b.toml b/providers/openrouter/models/qwen/qwen3-30b-a3b.toml new file mode 100644 index 000000000..eadb9141d --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-30b-a3b.toml @@ -0,0 +1,23 @@ +name = "Qwen3 30B A3B" +family = "qwen" +release_date = "2025-04-28" +last_updated = "2025-04-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.09 +output = 0.45 + +[limit] +context = 40_960 +output = 20_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-32b.toml b/providers/openrouter/models/qwen/qwen3-32b.toml new file mode 100644 index 000000000..dc5327343 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-32b.toml @@ -0,0 +1,23 @@ +name = "Qwen3 32B" +family = "qwen" +release_date = "2025-04-28" +last_updated = "2025-04-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.08 +output = 0.28 + +[limit] +context = 40_960 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-8b.toml b/providers/openrouter/models/qwen/qwen3-8b.toml new file mode 100644 index 000000000..9bbd1e585 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-8b.toml @@ -0,0 +1,24 @@ +name = "Qwen3 8B" +family = "qwen" +release_date = "2025-04-28" +last_updated = "2025-04-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.05 +output = 0.4 +cache_read = 0.05 + +[limit] +context = 40_960 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-coder-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-coder-30b-a3b-instruct.toml index 5b1b6b268..80d173c71 100644 --- a/providers/openrouter/models/qwen/qwen3-coder-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-coder-30b-a3b-instruct.toml @@ -5,9 +5,9 @@ last_updated = "2025-07-31" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = true [cost] @@ -16,7 +16,7 @@ output = 0.27 [limit] context = 160_000 -output = 65_536 +output = 32_768 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-coder-flash.toml b/providers/openrouter/models/qwen/qwen3-coder-flash.toml index 33df50a1c..ede004a6d 100644 --- a/providers/openrouter/models/qwen/qwen3-coder-flash.toml +++ b/providers/openrouter/models/qwen/qwen3-coder-flash.toml @@ -1,24 +1,24 @@ name = "Qwen3 Coder Flash" family = "qwen" -release_date = "2025-07-23" -last_updated = "2025-07-23" +release_date = "2025-09-17" +last_updated = "2025-09-17" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true -structured_output = false +structured_output = true +knowledge = "2025-06-30" open_weights = false [cost] -input = 0.30 -output = 1.50 +input = 0.195 +output = 0.975 cache_read = 0.039 cache_write = 0.24375 [limit] -context = 128_000 -output = 66_536 +context = 1_000_000 +output = 65_536 [modalities] input = ["text"] diff --git a/providers/openrouter/models/moonshotai/kimi-k2-0905:exacto.toml b/providers/openrouter/models/qwen/qwen3-coder-next.toml similarity index 53% rename from providers/openrouter/models/moonshotai/kimi-k2-0905:exacto.toml rename to providers/openrouter/models/qwen/qwen3-coder-next.toml index f44bfaf96..a7b9c5649 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2-0905:exacto.toml +++ b/providers/openrouter/models/qwen/qwen3-coder-next.toml @@ -1,22 +1,22 @@ -name = "Kimi K2 Instruct 0905 (exacto)" -family = "kimi" -release_date = "2025-09-05" -last_updated = "2025-09-05" +name = "Qwen3 Coder Next" +family = "qwen" +release_date = "2026-02-04" +last_updated = "2026-02-04" attachment = false reasoning = false temperature = true tool_call = true structured_output = true -knowledge = "2024-10" open_weights = true [cost] -input = 0.6 -output = 2.5 +input = 0.11 +output = 0.8 +cache_read = 0.07 [limit] context = 262_144 -output = 16_384 +output = 262_144 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-coder-plus.toml b/providers/openrouter/models/qwen/qwen3-coder-plus.toml index 256e93d37..ad18f7daa 100644 --- a/providers/openrouter/models/qwen/qwen3-coder-plus.toml +++ b/providers/openrouter/models/qwen/qwen3-coder-plus.toml @@ -5,10 +5,10 @@ last_updated = "2025-09-23" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true structured_output = true -open_weights = true +knowledge = "2025-06-30" +open_weights = false [cost] input = 0.65 diff --git a/providers/openrouter/models/qwen/qwen3-coder.toml b/providers/openrouter/models/qwen/qwen3-coder.toml index 8d35ff893..84d0fa703 100644 --- a/providers/openrouter/models/qwen/qwen3-coder.toml +++ b/providers/openrouter/models/qwen/qwen3-coder.toml @@ -1,23 +1,23 @@ -name = "Qwen3 Coder" +name = "Qwen3 Coder 480B A35B" family = "qwen" release_date = "2025-07-23" last_updated = "2025-07-23" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-06-30" open_weights = true [cost] -input = 0.3 -output = 1.2 +input = 0.22 +output = 1.8 [limit] context = 262_144 -output = 66_536 +output = 65_536 [modalities] input = ["text"] -output = ["text"] \ No newline at end of file +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-coder:exacto.toml b/providers/openrouter/models/qwen/qwen3-coder:free.toml similarity index 61% rename from providers/openrouter/models/qwen/qwen3-coder:exacto.toml rename to providers/openrouter/models/qwen/qwen3-coder:free.toml index 302b3e150..7562b381b 100644 --- a/providers/openrouter/models/qwen/qwen3-coder:exacto.toml +++ b/providers/openrouter/models/qwen/qwen3-coder:free.toml @@ -1,22 +1,22 @@ -name = "Qwen3 Coder (exacto)" +name = "Qwen3 Coder 480B A35B (free)" family = "qwen" release_date = "2025-07-23" last_updated = "2025-07-23" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true -structured_output = true +structured_output = false +knowledge = "2025-06-30" open_weights = true [cost] -input = 0.38 -output = 1.53 +input = 0 +output = 0 [limit] -context = 131_072 -output = 32_768 +context = 262_000 +output = 262_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/openrouter/elephant-alpha.toml b/providers/openrouter/models/qwen/qwen3-max-thinking.toml similarity index 63% rename from providers/openrouter/models/openrouter/elephant-alpha.toml rename to providers/openrouter/models/qwen/qwen3-max-thinking.toml index 36d16670b..4ed18ce91 100644 --- a/providers/openrouter/models/openrouter/elephant-alpha.toml +++ b/providers/openrouter/models/qwen/qwen3-max-thinking.toml @@ -1,7 +1,7 @@ -name = "Elephant (free)" -family = "elephant" -release_date = "2026-04-13" -last_updated = "2026-04-13" +name = "Qwen3 Max Thinking" +family = "qwen" +release_date = "2026-02-09" +last_updated = "2026-02-09" attachment = false reasoning = true temperature = true @@ -10,8 +10,8 @@ structured_output = true open_weights = false [cost] -input = 0.00 -output = 0.00 +input = 0.78 +output = 3.9 [limit] context = 262_144 diff --git a/providers/openrouter/models/qwen/qwen3-max.toml b/providers/openrouter/models/qwen/qwen3-max.toml index 34a838d26..65a31c98e 100644 --- a/providers/openrouter/models/qwen/qwen3-max.toml +++ b/providers/openrouter/models/qwen/qwen3-max.toml @@ -1,19 +1,18 @@ name = "Qwen3 Max" family = "qwen" -release_date = "2025-09-05" -last_updated = "2025-09-05" +release_date = "2025-09-23" +last_updated = "2025-09-23" attachment = false -reasoning = true +reasoning = false temperature = true tool_call = true +structured_output = true +knowledge = "2025-06-30" open_weights = false [cost] -# cost at >= 128k context -# input = 1 -# output = 5 -input = 1.20 -output = 6.00 +input = 0.78 +output = 3.9 cache_read = 0.156 cache_write = 0.975 diff --git a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct.toml index bddd233ae..00483c4c9 100644 --- a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct.toml @@ -5,18 +5,18 @@ last_updated = "2025-09-11" attachment = false reasoning = false temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-09-30" open_weights = true [cost] -input = 0.14 -output = 1.40 +input = 0.09 +output = 1.1 [limit] context = 262_144 -output = 262_144 +output = 16_384 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct:free.toml b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct:free.toml new file mode 100644 index 000000000..eb683d5a1 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-instruct:free.toml @@ -0,0 +1,23 @@ +name = "Qwen3 Next 80B A3B Instruct (free)" +family = "qwen" +release_date = "2025-09-11" +last_updated = "2025-09-11" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-09-30" +open_weights = true + +[cost] +input = 0 +output = 0 + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml index b4a575fe3..9ed138ba8 100644 --- a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml +++ b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml @@ -5,18 +5,18 @@ last_updated = "2025-09-11" attachment = false reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-09-30" open_weights = true [cost] -input = 0.14 -output = 1.40 +input = 0.0975 +output = 0.78 [limit] -context = 262_144 -output = 262_144 +context = 131_072 +output = 32_768 [modalities] input = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-235b-a22b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-235b-a22b-instruct.toml new file mode 100644 index 000000000..2d1ae95fb --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-235b-a22b-instruct.toml @@ -0,0 +1,24 @@ +name = "Qwen3 VL 235B A22B Instruct" +family = "qwen" +release_date = "2025-09-23" +last_updated = "2025-09-23" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.2 +output = 0.88 +cache_read = 0.11 + +[limit] +context = 262_144 +output = 16_384 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-235b-a22b-thinking.toml b/providers/openrouter/models/qwen/qwen3-vl-235b-a22b-thinking.toml new file mode 100644 index 000000000..9a4d1d859 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-235b-a22b-thinking.toml @@ -0,0 +1,23 @@ +name = "Qwen3 VL 235B A22B Thinking" +family = "qwen" +release_date = "2025-09-23" +last_updated = "2025-09-23" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.26 +output = 2.6 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml new file mode 100644 index 000000000..c8c272429 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -0,0 +1,23 @@ +name = "Qwen3 VL 30B A3B Instruct" +family = "qwen" +release_date = "2025-10-06" +last_updated = "2025-10-06" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.13 +output = 0.52 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-thinking.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-thinking.toml new file mode 100644 index 000000000..f1727b17c --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-thinking.toml @@ -0,0 +1,23 @@ +name = "Qwen3 VL 30B A3B Thinking" +family = "qwen" +release_date = "2025-10-06" +last_updated = "2025-10-06" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.13 +output = 1.56 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-32b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-32b-instruct.toml new file mode 100644 index 000000000..5250189e7 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-32b-instruct.toml @@ -0,0 +1,22 @@ +name = "Qwen3 VL 32B Instruct" +family = "qwen" +release_date = "2025-10-23" +last_updated = "2025-10-23" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.104 +output = 0.416 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-8b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-8b-instruct.toml new file mode 100644 index 000000000..dac25c070 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-8b-instruct.toml @@ -0,0 +1,22 @@ +name = "Qwen3 VL 8B Instruct" +family = "qwen" +release_date = "2025-10-14" +last_updated = "2025-10-14" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.08 +output = 0.5 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-vl-8b-thinking.toml b/providers/openrouter/models/qwen/qwen3-vl-8b-thinking.toml new file mode 100644 index 000000000..dff64189e --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3-vl-8b-thinking.toml @@ -0,0 +1,22 @@ +name = "Qwen3 VL 8B Thinking" +family = "qwen" +release_date = "2025-10-14" +last_updated = "2025-10-14" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.117 +output = 1.365 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.5-122b-a10b.toml b/providers/openrouter/models/qwen/qwen3.5-122b-a10b.toml new file mode 100644 index 000000000..6f6bd2b9f --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.5-122b-a10b.toml @@ -0,0 +1,22 @@ +name = "Qwen3.5-122B-A10B" +family = "qwen3.5" +release_date = "2026-02-25" +last_updated = "2026-02-25" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.26 +output = 2.08 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.5-27b.toml b/providers/openrouter/models/qwen/qwen3.5-27b.toml new file mode 100644 index 000000000..a3c6b6eb5 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.5-27b.toml @@ -0,0 +1,22 @@ +name = "Qwen3.5-27B" +family = "qwen3.5" +release_date = "2026-02-25" +last_updated = "2026-02-25" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.195 +output = 1.56 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml b/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml new file mode 100644 index 000000000..68120862f --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml @@ -0,0 +1,23 @@ +name = "Qwen3.5-35B-A3B" +family = "qwen3.5" +release_date = "2026-02-25" +last_updated = "2026-02-25" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.14 +output = 1 +cache_read = 0.05 + +[limit] +context = 262_144 +output = 81_920 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.5-397b-a17b.toml b/providers/openrouter/models/qwen/qwen3.5-397b-a17b.toml index 781d7983d..97c5a09ac 100644 --- a/providers/openrouter/models/qwen/qwen3.5-397b-a17b.toml +++ b/providers/openrouter/models/qwen/qwen3.5-397b-a17b.toml @@ -5,14 +5,15 @@ last_updated = "2026-02-16" attachment = true reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-04" open_weights = true [cost] -input = 0.60 -output = 3.60 +input = 0.39 +output = 2.34 +cache_read = 0.195 [limit] context = 262_144 diff --git a/providers/openrouter/models/qwen/qwen3.5-9b.toml b/providers/openrouter/models/qwen/qwen3.5-9b.toml new file mode 100644 index 000000000..0623c88fd --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.5-9b.toml @@ -0,0 +1,22 @@ +name = "Qwen3.5-9B" +family = "qwen3.5" +release_date = "2026-03-10" +last_updated = "2026-03-10" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.04 +output = 0.15 + +[limit] +context = 262_144 +output = 81_920 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.5-flash-02-23.toml b/providers/openrouter/models/qwen/qwen3.5-flash-02-23.toml index 1e967179c..cc2a78eb2 100644 --- a/providers/openrouter/models/qwen/qwen3.5-flash-02-23.toml +++ b/providers/openrouter/models/qwen/qwen3.5-flash-02-23.toml @@ -1,4 +1,4 @@ -name = "Qwen: Qwen3.5-Flash" +name = "Qwen3.5-Flash" family = "qwen" release_date = "2026-02-25" last_updated = "2026-02-25" @@ -12,6 +12,7 @@ open_weights = false [cost] input = 0.065 output = 0.26 +cache_write = 0.08125 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/qwen/qwen3.5-plus-02-15.toml b/providers/openrouter/models/qwen/qwen3.5-plus-02-15.toml index 962a26751..99a53635d 100644 --- a/providers/openrouter/models/qwen/qwen3.5-plus-02-15.toml +++ b/providers/openrouter/models/qwen/qwen3.5-plus-02-15.toml @@ -5,14 +5,15 @@ last_updated = "2026-02-16" attachment = true reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-04" open_weights = false [cost] -input = 0.40 -output = 2.40 +input = 0.26 +output = 1.56 +cache_write = 0.325 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/qwen/qwen3.5-plus-20260420.toml b/providers/openrouter/models/qwen/qwen3.5-plus-20260420.toml new file mode 100644 index 000000000..30f3655ad --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.5-plus-20260420.toml @@ -0,0 +1,22 @@ +name = "Qwen3.5 Plus 2026-04-20" +family = "qwen3.5" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.3 +output = 1.8 + +[limit] +context = 1_000_000 +output = 65_536 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen-3.6-27b.toml b/providers/openrouter/models/qwen/qwen3.6-27b.toml similarity index 59% rename from providers/openrouter/models/qwen/qwen-3.6-27b.toml rename to providers/openrouter/models/qwen/qwen3.6-27b.toml index ff70981fb..668fd2573 100644 --- a/providers/openrouter/models/qwen/qwen-3.6-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.6-27b.toml @@ -1,19 +1,17 @@ -id = "qwen/qwen3.6-27b" name = "Qwen3.6 27B" -family = "qwen" -release_date = "2026-04-22" -last_updated = "2026-04-22" +family = "qwen3.6" +release_date = "2026-04-27" +last_updated = "2026-04-27" attachment = true -reasoning = false +reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true open_weights = true [cost] -input = 0.195 -output = 1.56 +input = 0.32 +output = 3.2 [limit] context = 262_144 diff --git a/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml b/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml new file mode 100644 index 000000000..7d876ca0d --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml @@ -0,0 +1,23 @@ +name = "Qwen3.6 35B A3B" +family = "qwen3.6" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.15 +output = 1 +cache_read = 0.05 + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.6-flash.toml b/providers/openrouter/models/qwen/qwen3.6-flash.toml new file mode 100644 index 000000000..9ff48336d --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.6-flash.toml @@ -0,0 +1,23 @@ +name = "Qwen3.6 Flash" +family = "qwen3.6" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.1875 +output = 1.125 +cache_write = 0.234375 + +[limit] +context = 1_000_000 +output = 65_536 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.6-max-preview.toml b/providers/openrouter/models/qwen/qwen3.6-max-preview.toml new file mode 100644 index 000000000..3775091f6 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.6-max-preview.toml @@ -0,0 +1,23 @@ +name = "Qwen3.6 Max Preview" +family = "qwen3.6" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 1.04 +output = 6.24 +cache_write = 1.3 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3.6-plus.toml b/providers/openrouter/models/qwen/qwen3.6-plus.toml index 57898dcbe..46cf70cce 100644 --- a/providers/openrouter/models/qwen/qwen3.6-plus.toml +++ b/providers/openrouter/models/qwen/qwen3.6-plus.toml @@ -5,15 +5,14 @@ last_updated = "2026-04-02" attachment = true reasoning = true temperature = true -knowledge = "2025-04" tool_call = true structured_output = true +knowledge = "2025-04" open_weights = false [cost] input = 0.325 output = 1.95 -cache_read = 0.0325 cache_write = 0.40625 [limit] diff --git a/providers/openrouter/models/rekaai/reka-edge.toml b/providers/openrouter/models/rekaai/reka-edge.toml new file mode 100644 index 000000000..096c5b9bd --- /dev/null +++ b/providers/openrouter/models/rekaai/reka-edge.toml @@ -0,0 +1,22 @@ +name = "Reka Edge" +family = "reka" +release_date = "2026-03-20" +last_updated = "2026-03-20" +attachment = true +reasoning = false +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.1 +output = 0.1 + +[limit] +context = 16_384 +output = 16_384 + +[modalities] +input = ["image", "text", "video"] +output = ["text"] diff --git a/providers/openrouter/models/rekaai/reka-flash-3.toml b/providers/openrouter/models/rekaai/reka-flash-3.toml new file mode 100644 index 000000000..66f4c71d5 --- /dev/null +++ b/providers/openrouter/models/rekaai/reka-flash-3.toml @@ -0,0 +1,23 @@ +name = "Reka Flash 3" +family = "reka" +release_date = "2025-03-12" +last_updated = "2025-03-12" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +knowledge = "2025-01-31" +open_weights = true + +[cost] +input = 0.1 +output = 0.2 + +[limit] +context = 65_536 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/relace/relace-apply-3.toml b/providers/openrouter/models/relace/relace-apply-3.toml new file mode 100644 index 000000000..e64a0ff8c --- /dev/null +++ b/providers/openrouter/models/relace/relace-apply-3.toml @@ -0,0 +1,21 @@ +name = "Relace Apply 3" +release_date = "2025-09-26" +last_updated = "2025-09-26" +attachment = false +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.85 +output = 1.25 + +[limit] +context = 256_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/relace/relace-search.toml b/providers/openrouter/models/relace/relace-search.toml new file mode 100644 index 000000000..2c4ac1fb1 --- /dev/null +++ b/providers/openrouter/models/relace/relace-search.toml @@ -0,0 +1,21 @@ +name = "Relace Search" +release_date = "2025-12-08" +last_updated = "2025-12-08" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[cost] +input = 1 +output = 3 + +[limit] +context = 256_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/sao10k/l3-euryale-70b.toml b/providers/openrouter/models/sao10k/l3-euryale-70b.toml new file mode 100644 index 000000000..a3adb53b6 --- /dev/null +++ b/providers/openrouter/models/sao10k/l3-euryale-70b.toml @@ -0,0 +1,23 @@ +name = "Llama 3 Euryale 70B v2.1" +family = "llama" +release_date = "2024-06-18" +last_updated = "2024-06-18" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 1.48 +output = 1.48 + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/sao10k/l3-lunaris-8b.toml b/providers/openrouter/models/sao10k/l3-lunaris-8b.toml new file mode 100644 index 000000000..19c8433af --- /dev/null +++ b/providers/openrouter/models/sao10k/l3-lunaris-8b.toml @@ -0,0 +1,23 @@ +name = "Llama 3 8B Lunaris" +family = "llama" +release_date = "2024-08-13" +last_updated = "2024-08-13" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.04 +output = 0.05 + +[limit] +context = 8_192 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/sao10k/l3.1-70b-hanami-x1.toml b/providers/openrouter/models/sao10k/l3.1-70b-hanami-x1.toml new file mode 100644 index 000000000..a23e5a1a3 --- /dev/null +++ b/providers/openrouter/models/sao10k/l3.1-70b-hanami-x1.toml @@ -0,0 +1,23 @@ +name = "Llama 3.1 70B Hanami x1" +family = "llama" +release_date = "2025-01-08" +last_updated = "2025-01-08" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 3 +output = 3 + +[limit] +context = 16_000 +output = 16_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/sao10k/l3.1-euryale-70b.toml b/providers/openrouter/models/sao10k/l3.1-euryale-70b.toml new file mode 100644 index 000000000..058769c28 --- /dev/null +++ b/providers/openrouter/models/sao10k/l3.1-euryale-70b.toml @@ -0,0 +1,23 @@ +name = "Llama 3.1 Euryale 70B v2.2" +family = "llama" +release_date = "2024-08-28" +last_updated = "2024-08-28" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.85 +output = 0.85 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/sao10k/l3.3-euryale-70b.toml b/providers/openrouter/models/sao10k/l3.3-euryale-70b.toml new file mode 100644 index 000000000..11a2c9457 --- /dev/null +++ b/providers/openrouter/models/sao10k/l3.3-euryale-70b.toml @@ -0,0 +1,23 @@ +name = "Llama 3.3 Euryale 70B" +family = "llama" +release_date = "2024-12-18" +last_updated = "2024-12-18" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-12-31" +open_weights = true + +[cost] +input = 0.65 +output = 0.75 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/sourceful/riverflow-v2-fast-preview.toml b/providers/openrouter/models/sourceful/riverflow-v2-fast-preview.toml deleted file mode 100644 index e3ce1a096..000000000 --- a/providers/openrouter/models/sourceful/riverflow-v2-fast-preview.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Riverflow V2 Fast Preview" -family = "sourceful" -release_date = "2025-12-08" -last_updated = "2026-01-28" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 8_192 -output = 8_192 - -[modalities] -input = ["text","image"] -output = ["image"] diff --git a/providers/openrouter/models/sourceful/riverflow-v2-max-preview.toml b/providers/openrouter/models/sourceful/riverflow-v2-max-preview.toml deleted file mode 100644 index f7cc7f793..000000000 --- a/providers/openrouter/models/sourceful/riverflow-v2-max-preview.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Riverflow V2 Max Preview" -family = "sourceful" -release_date = "2025-12-08" -last_updated = "2026-01-28" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 8_192 -output = 8_192 - -[modalities] -input = ["text","image"] -output = ["image"] diff --git a/providers/openrouter/models/sourceful/riverflow-v2-standard-preview.toml b/providers/openrouter/models/sourceful/riverflow-v2-standard-preview.toml deleted file mode 100644 index 575871174..000000000 --- a/providers/openrouter/models/sourceful/riverflow-v2-standard-preview.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Riverflow V2 Standard Preview" -family = "sourceful" -release_date = "2025-12-08" -last_updated = "2026-01-28" -attachment = false -reasoning = false -temperature = true -# may be inaccurate -knowledge = "2025-06" -tool_call = false -open_weights = true - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 8_192 -output = 8_192 - -[modalities] -input = ["text","image"] -output = ["image"] diff --git a/providers/openrouter/models/stepfun/step-3.5-flash.toml b/providers/openrouter/models/stepfun/step-3.5-flash.toml index 8d526d7b3..5ecbbfbec 100644 --- a/providers/openrouter/models/stepfun/step-3.5-flash.toml +++ b/providers/openrouter/models/stepfun/step-3.5-flash.toml @@ -6,17 +6,17 @@ attachment = false reasoning = true temperature = true tool_call = true +structured_output = true knowledge = "2025-01" open_weights = true [cost] -input = 0.10 -output = 0.30 -cache_read = 0.02 +input = 0.1 +output = 0.3 [limit] -context = 256_000 -output = 256_000 +context = 262_144 +output = 65_536 [modalities] input = ["text"] diff --git a/providers/openrouter/models/switchpoint/router.toml b/providers/openrouter/models/switchpoint/router.toml new file mode 100644 index 000000000..e4411535d --- /dev/null +++ b/providers/openrouter/models/switchpoint/router.toml @@ -0,0 +1,22 @@ +name = "Switchpoint Router" +family = "o" +release_date = "2025-07-11" +last_updated = "2025-07-11" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.85 +output = 3.4 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/tencent/hunyuan-a13b-instruct.toml b/providers/openrouter/models/tencent/hunyuan-a13b-instruct.toml new file mode 100644 index 000000000..872a320ba --- /dev/null +++ b/providers/openrouter/models/tencent/hunyuan-a13b-instruct.toml @@ -0,0 +1,23 @@ +name = "Hunyuan A13B Instruct" +family = "hunyuan" +release_date = "2025-07-08" +last_updated = "2025-07-08" +attachment = false +reasoning = true +temperature = true +tool_call = false +structured_output = true +knowledge = "2025-03-31" +open_weights = true + +[cost] +input = 0.14 +output = 0.57 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/tencent/hy3-preview.toml b/providers/openrouter/models/tencent/hy3-preview.toml index c1a97afb4..d51ebb8a1 100644 --- a/providers/openrouter/models/tencent/hy3-preview.toml +++ b/providers/openrouter/models/tencent/hy3-preview.toml @@ -1,22 +1,22 @@ name = "Hy3 preview" family = "Hy" -release_date = "2026-04-20" -last_updated = "2026-04-20" +release_date = "2026-04-22" +last_updated = "2026-04-22" attachment = false reasoning = true -tool_call = true temperature = true +tool_call = true +structured_output = false open_weights = true [cost] input = 0.066 output = 0.26 cache_read = 0.029 -cache_write = 0.029 [limit] -context = 256000 -output = 64000 +context = 262_144 +output = 262_144 [modalities] input = ["text"] diff --git a/providers/openrouter/models/thedrummer/cydonia-24b-v4.1.toml b/providers/openrouter/models/thedrummer/cydonia-24b-v4.1.toml new file mode 100644 index 000000000..cdeee01d3 --- /dev/null +++ b/providers/openrouter/models/thedrummer/cydonia-24b-v4.1.toml @@ -0,0 +1,24 @@ +name = "Cydonia 24B V4.1" +family = "o" +release_date = "2025-09-27" +last_updated = "2025-09-27" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2024-04-30" +open_weights = true + +[cost] +input = 0.3 +output = 0.5 +cache_read = 0.15 + +[limit] +context = 131_072 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/thedrummer/rocinante-12b.toml b/providers/openrouter/models/thedrummer/rocinante-12b.toml new file mode 100644 index 000000000..db6d95fce --- /dev/null +++ b/providers/openrouter/models/thedrummer/rocinante-12b.toml @@ -0,0 +1,23 @@ +name = "Rocinante 12B" +family = "o" +release_date = "2024-09-30" +last_updated = "2024-09-30" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-04-30" +open_weights = true + +[cost] +input = 0.17 +output = 0.43 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/thedrummer/skyfall-36b-v2.toml b/providers/openrouter/models/thedrummer/skyfall-36b-v2.toml new file mode 100644 index 000000000..2f17e0bc6 --- /dev/null +++ b/providers/openrouter/models/thedrummer/skyfall-36b-v2.toml @@ -0,0 +1,23 @@ +name = "Skyfall 36B V2" +release_date = "2025-03-10" +last_updated = "2025-03-10" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +knowledge = "2024-06-30" +open_weights = true + +[cost] +input = 0.55 +output = 0.8 +cache_read = 0.25 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/thedrummer/unslopnemo-12b.toml b/providers/openrouter/models/thedrummer/unslopnemo-12b.toml new file mode 100644 index 000000000..2c1c923b4 --- /dev/null +++ b/providers/openrouter/models/thedrummer/unslopnemo-12b.toml @@ -0,0 +1,23 @@ +name = "UnslopNemo 12B" +family = "o" +release_date = "2024-11-08" +last_updated = "2024-11-08" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-04-30" +open_weights = true + +[cost] +input = 0.4 +output = 0.4 + +[limit] +context = 32_768 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/undi95/remm-slerp-l2-13b.toml b/providers/openrouter/models/undi95/remm-slerp-l2-13b.toml new file mode 100644 index 000000000..acee8f947 --- /dev/null +++ b/providers/openrouter/models/undi95/remm-slerp-l2-13b.toml @@ -0,0 +1,22 @@ +name = "ReMM SLERP 13B" +release_date = "2023-07-22" +last_updated = "2023-07-22" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = true +knowledge = "2023-06-30" +open_weights = true + +[cost] +input = 0.45 +output = 0.65 + +[limit] +context = 6_144 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/upstage/solar-pro-3.toml b/providers/openrouter/models/upstage/solar-pro-3.toml new file mode 100644 index 000000000..22f35184b --- /dev/null +++ b/providers/openrouter/models/upstage/solar-pro-3.toml @@ -0,0 +1,23 @@ +name = "Solar Pro 3" +family = "solar-pro" +release_date = "2026-01-27" +last_updated = "2026-01-27" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.015 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/writer/palmyra-x5.toml b/providers/openrouter/models/writer/palmyra-x5.toml new file mode 100644 index 000000000..779b08e40 --- /dev/null +++ b/providers/openrouter/models/writer/palmyra-x5.toml @@ -0,0 +1,22 @@ +name = "Palmyra X5" +family = "palmyra" +release_date = "2026-01-21" +last_updated = "2026-01-21" +attachment = false +reasoning = false +temperature = true +tool_call = false +structured_output = false +open_weights = false + +[cost] +input = 0.6 +output = 6 + +[limit] +context = 1_040_000 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-3-beta.toml b/providers/openrouter/models/x-ai/grok-3-beta.toml index 2f7eed685..3bb6d214b 100644 --- a/providers/openrouter/models/x-ai/grok-3-beta.toml +++ b/providers/openrouter/models/x-ai/grok-3-beta.toml @@ -1,19 +1,19 @@ name = "Grok 3 Beta" family = "grok" -release_date = "2025-02-17" -last_updated = "2025-02-17" +release_date = "2025-04-09" +last_updated = "2025-04-09" attachment = false reasoning = false temperature = true -knowledge = "2024-11" tool_call = true +structured_output = true +knowledge = "2025-02-28" open_weights = false [cost] -input = 3.00 -output = 15.00 +input = 3 +output = 15 cache_read = 0.75 -cache_write = 15.00 [limit] context = 131_072 diff --git a/providers/openrouter/models/x-ai/grok-3-mini-beta.toml b/providers/openrouter/models/x-ai/grok-3-mini-beta.toml index 06c4d8f67..0896dd2bf 100644 --- a/providers/openrouter/models/x-ai/grok-3-mini-beta.toml +++ b/providers/openrouter/models/x-ai/grok-3-mini-beta.toml @@ -1,19 +1,19 @@ name = "Grok 3 Mini Beta" family = "grok" -release_date = "2025-02-17" -last_updated = "2025-02-17" +release_date = "2025-04-09" +last_updated = "2025-04-09" attachment = false reasoning = true temperature = true -knowledge = "2024-11" tool_call = true +structured_output = true +knowledge = "2025-02-28" open_weights = false [cost] -input = 0.30 -output = 0.50 +input = 0.3 +output = 0.5 cache_read = 0.075 -cache_write = 0.50 [limit] context = 131_072 diff --git a/providers/openrouter/models/x-ai/grok-3-mini.toml b/providers/openrouter/models/x-ai/grok-3-mini.toml index ba1b712dc..4005bcc0e 100644 --- a/providers/openrouter/models/x-ai/grok-3-mini.toml +++ b/providers/openrouter/models/x-ai/grok-3-mini.toml @@ -1,20 +1,19 @@ name = "Grok 3 Mini" family = "grok" -release_date = "2025-02-17" -last_updated = "2025-02-17" +release_date = "2025-06-10" +last_updated = "2025-06-10" attachment = false reasoning = true temperature = true -knowledge = "2024-11" tool_call = true structured_output = true +knowledge = "2025-02-28" open_weights = false [cost] -input = 0.30 -output = 0.50 +input = 0.3 +output = 0.5 cache_read = 0.075 -cache_write = 0.50 [limit] context = 131_072 diff --git a/providers/openrouter/models/x-ai/grok-3.toml b/providers/openrouter/models/x-ai/grok-3.toml index 8ba0679d4..40f980b50 100644 --- a/providers/openrouter/models/x-ai/grok-3.toml +++ b/providers/openrouter/models/x-ai/grok-3.toml @@ -1,20 +1,19 @@ name = "Grok 3" family = "grok" -release_date = "2025-02-17" -last_updated = "2025-02-17" +release_date = "2025-06-10" +last_updated = "2025-06-10" attachment = false reasoning = false temperature = true -knowledge = "2024-11" tool_call = true structured_output = true +knowledge = "2025-02-28" open_weights = false [cost] -input = 3.00 -output = 15.00 +input = 3 +output = 15 cache_read = 0.75 -cache_write = 15.00 [limit] context = 131_072 diff --git a/providers/openrouter/models/x-ai/grok-4-fast.toml b/providers/openrouter/models/x-ai/grok-4-fast.toml index 09da5927e..47abebfd8 100644 --- a/providers/openrouter/models/x-ai/grok-4-fast.toml +++ b/providers/openrouter/models/x-ai/grok-4-fast.toml @@ -1,25 +1,24 @@ name = "Grok 4 Fast" family = "grok" -release_date = "2025-08-19" -last_updated = "2025-08-19" -attachment = false +release_date = "2025-09-19" +last_updated = "2025-09-19" +attachment = true reasoning = true temperature = true -knowledge = "2024-11" tool_call = true structured_output = true +knowledge = "2025-09-30" open_weights = false [cost] -input = 0.20 -output = 0.50 +input = 0.2 +output = 0.5 cache_read = 0.05 -cache_write = 0.05 [limit] context = 2_000_000 output = 30_000 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.1-fast.toml b/providers/openrouter/models/x-ai/grok-4.1-fast.toml index f1660a8ea..8bcab1305 100644 --- a/providers/openrouter/models/x-ai/grok-4.1-fast.toml +++ b/providers/openrouter/models/x-ai/grok-4.1-fast.toml @@ -2,24 +2,23 @@ name = "Grok 4.1 Fast" family = "grok" release_date = "2025-11-19" last_updated = "2025-11-19" -attachment = false +attachment = true reasoning = true temperature = true -knowledge = "2024-11" tool_call = true structured_output = true +knowledge = "2024-11" open_weights = false [cost] -input = 0.20 -output = 0.50 +input = 0.2 +output = 0.5 cache_read = 0.05 -cache_write = 0.05 [limit] context = 2_000_000 output = 30_000 [modalities] -input = ["text", "image"] -output = ["text"] \ No newline at end of file +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.20-beta.toml b/providers/openrouter/models/x-ai/grok-4.20-beta.toml deleted file mode 100644 index dae021999..000000000 --- a/providers/openrouter/models/x-ai/grok-4.20-beta.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Grok 4.20 Beta" -family = "grok" -status = "beta" -release_date = "2026-03-12" -last_updated = "2026-03-12" -attachment = true -reasoning = true -temperature = true -tool_call = true -open_weights = false - -[cost] -input = 2.00 -output = 6.00 -cache_read = 0.20 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 4.00 -output = 12.00 -cache_read = 0.40 - -[limit] -context = 2_000_000 -output = 30_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.20-multi-agent-beta.toml b/providers/openrouter/models/x-ai/grok-4.20-multi-agent-beta.toml deleted file mode 100644 index ddc403197..000000000 --- a/providers/openrouter/models/x-ai/grok-4.20-multi-agent-beta.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Grok 4.20 Multi - Agent Beta" -family = "grok" -status = "beta" -release_date = "2026-03-12" -last_updated = "2026-03-12" -attachment = true -reasoning = true -temperature = true -tool_call = false -open_weights = false - -[cost] -input = 2.00 -output = 6.00 -cache_read = 0.20 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 4.00 -output = 12.00 -cache_read = 0.40 - -[limit] -context = 2_000_000 -output = 30_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.20-multi-agent.toml b/providers/openrouter/models/x-ai/grok-4.20-multi-agent.toml new file mode 100644 index 000000000..ff1ba4a0b --- /dev/null +++ b/providers/openrouter/models/x-ai/grok-4.20-multi-agent.toml @@ -0,0 +1,24 @@ +name = "Grok 4.20 Multi-Agent" +family = "grok" +release_date = "2026-03-31" +last_updated = "2026-03-31" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +knowledge = "2025-09-01" +open_weights = false + +[cost] +input = 2 +output = 6 +cache_read = 0.2 + +[limit] +context = 2_000_000 +output = 2_000_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.20.toml b/providers/openrouter/models/x-ai/grok-4.20.toml new file mode 100644 index 000000000..c1a84f136 --- /dev/null +++ b/providers/openrouter/models/x-ai/grok-4.20.toml @@ -0,0 +1,24 @@ +name = "Grok 4.20" +family = "grok" +release_date = "2026-03-31" +last_updated = "2026-03-31" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-09-01" +open_weights = false + +[cost] +input = 1.25 +output = 2.5 +cache_read = 0.2 + +[limit] +context = 2_000_000 +output = 2_000_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.3.toml b/providers/openrouter/models/x-ai/grok-4.3.toml index de45d4744..2d596af84 100644 --- a/providers/openrouter/models/x-ai/grok-4.3.toml +++ b/providers/openrouter/models/x-ai/grok-4.3.toml @@ -1,19 +1,29 @@ +name = "Grok 4.3" +family = "grok" +release_date = "2026-04-30" +last_updated = "2026-04-30" +attachment = true +reasoning = true +temperature = true +tool_call = true structured_output = true - -[extends] -from = "xai/grok-4.3" +open_weights = false [cost] input = 1.25 -output = 2.50 -cache_read = 0.20 +output = 2.5 +cache_read = 0.2 [[cost.tiers]] tier = { size = 200_000 } -input = 2.50 -output = 5.00 -cache_read = 0.40 +input = 2.5 +output = 5 +cache_read = 0.4 [limit] context = 1_000_000 output = 1_000_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-4.toml b/providers/openrouter/models/x-ai/grok-4.toml index 721307358..70e58c3ea 100644 --- a/providers/openrouter/models/x-ai/grok-4.toml +++ b/providers/openrouter/models/x-ai/grok-4.toml @@ -2,29 +2,28 @@ name = "Grok 4" family = "grok" release_date = "2025-07-09" last_updated = "2025-07-09" -attachment = false +attachment = true reasoning = true temperature = true -knowledge = "2025-07" tool_call = true structured_output = true +knowledge = "2025-07-31" open_weights = false [cost] -input = 3.00 -output = 15.00 +input = 3 +output = 15 cache_read = 0.75 -cache_write = 15.00 [[cost.tiers]] tier = { size = 128_000 } -input = 6.00 -output = 30.00 +input = 6 +output = 30 [limit] context = 256_000 output = 64_000 [modalities] -input = ["text"] +input = ["image", "text", "pdf"] output = ["text"] diff --git a/providers/openrouter/models/x-ai/grok-code-fast-1.toml b/providers/openrouter/models/x-ai/grok-code-fast-1.toml index 94777f663..4cb5b6113 100644 --- a/providers/openrouter/models/x-ai/grok-code-fast-1.toml +++ b/providers/openrouter/models/x-ai/grok-code-fast-1.toml @@ -5,14 +5,14 @@ last_updated = "2025-08-26" attachment = false reasoning = true temperature = true -knowledge = "2025-08" tool_call = true structured_output = true +knowledge = "2025-09-30" open_weights = false [cost] -input = 0.20 -output = 1.50 +input = 0.2 +output = 1.5 cache_read = 0.02 [limit] diff --git a/providers/openrouter/models/xiaomi/mimo-v2-flash.toml b/providers/openrouter/models/xiaomi/mimo-v2-flash.toml index 4b61f0544..8e4851667 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2-flash.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2-flash.toml @@ -1,7 +1,26 @@ -name = "Xiaomi: MiMo-V2-Flash" - -[extends] -from = "xiaomi/mimo-v2-flash" +name = "MiMo-V2-Flash" +family = "mimo" +release_date = "2025-12-14" +last_updated = "2025-12-14" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true [interleaved] field = "reasoning_details" + +[cost] +input = 0.1 +output = 0.3 +cache_read = 0.01 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2-omni.toml b/providers/openrouter/models/xiaomi/mimo-v2-omni.toml index 0d5526987..4a929f5b1 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2-omni.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2-omni.toml @@ -1,7 +1,26 @@ -name = "Xiaomi: MiMo-V2-Omni" - -[extends] -from = "xiaomi/mimo-v2-omni" +name = "MiMo-V2-Omni" +family = "mimo-v2-omni" +release_date = "2026-03-18" +last_updated = "2026-03-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false [interleaved] field = "reasoning_details" + +[cost] +input = 0.4 +output = 2 +cache_read = 0.08 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text", "audio", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2-pro.toml b/providers/openrouter/models/xiaomi/mimo-v2-pro.toml index d80f0ff85..f6861d0f8 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2-pro.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2-pro.toml @@ -1,7 +1,26 @@ -name = "Xiaomi: MiMo-V2-Pro" - -[extends] -from = "xiaomi/mimo-v2-pro" +name = "MiMo-V2-Pro" +family = "mimo-v2-pro" +release_date = "2026-03-18" +last_updated = "2026-03-18" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false [interleaved] field = "reasoning_details" + +[cost] +input = 1 +output = 3 +cache_read = 0.2 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2.5-pro.toml b/providers/openrouter/models/xiaomi/mimo-v2.5-pro.toml index 9b422ef8c..7726dd186 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2.5-pro.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2.5-pro.toml @@ -1,4 +1,23 @@ -name = "Xiaomi: MiMo-V2.5-Pro" +name = "MiMo-V2.5-Pro" +family = "mimo-v2.5-pro" +release_date = "2026-04-22" +last_updated = "2026-04-22" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true -[extends] -from = "xiaomi/mimo-v2.5-pro" +[cost] +input = 1 +output = 3 +cache_read = 0.2 + +[limit] +context = 1_048_576 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2.5.toml b/providers/openrouter/models/xiaomi/mimo-v2.5.toml index 8e0a1a5a7..88487a1a2 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2.5.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2.5.toml @@ -1,7 +1,26 @@ -name = "Xiaomi: MiMo-V2.5" - -[extends] -from = "xiaomi/mimo-v2.5" +name = "MiMo-V2.5" +family = "mimo-v2.5" +release_date = "2026-04-22" +last_updated = "2026-04-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true [interleaved] field = "reasoning_details" + +[cost] +input = 0.4 +output = 2 +cache_read = 0.08 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "audio", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4-32b.toml b/providers/openrouter/models/z-ai/glm-4-32b.toml new file mode 100644 index 000000000..ffc77edb4 --- /dev/null +++ b/providers/openrouter/models/z-ai/glm-4-32b.toml @@ -0,0 +1,23 @@ +name = "GLM 4 32B " +family = "glm" +release_date = "2025-07-24" +last_updated = "2025-07-24" +attachment = false +reasoning = false +temperature = true +tool_call = true +structured_output = false +knowledge = "2024-06-30" +open_weights = false + +[cost] +input = 0.1 +output = 0.1 + +[limit] +context = 128_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.5-air.toml b/providers/openrouter/models/z-ai/glm-4.5-air.toml index 4960ee095..6ba0291ff 100644 --- a/providers/openrouter/models/z-ai/glm-4.5-air.toml +++ b/providers/openrouter/models/z-ai/glm-4.5-air.toml @@ -1,22 +1,23 @@ name = "GLM 4.5 Air" family = "glm-air" -release_date = "2025-07-28" -last_updated = "2025-07-28" +release_date = "2025-07-25" +last_updated = "2025-07-25" attachment = false reasoning = true temperature = true tool_call = true -structured_output = true -knowledge = "2025-04" +structured_output = false +knowledge = "2024-12-31" open_weights = true [cost] -input = 0.20 -output = 1.10 +input = 0.13 +output = 0.85 +cache_read = 0.025 [limit] -context = 128_000 -output = 96_000 +context = 131_072 +output = 98_304 [modalities] input = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.5-air:free.toml b/providers/openrouter/models/z-ai/glm-4.5-air:free.toml index d33541d64..94ebd791c 100644 --- a/providers/openrouter/models/z-ai/glm-4.5-air:free.toml +++ b/providers/openrouter/models/z-ai/glm-4.5-air:free.toml @@ -1,20 +1,21 @@ name = "GLM 4.5 Air (free)" family = "glm-air" -release_date = "2025-07-28" -last_updated = "2025-07-28" +release_date = "2025-07-25" +last_updated = "2025-07-25" attachment = false reasoning = true temperature = true -tool_call = false -knowledge = "2025-04" +tool_call = true +structured_output = false +knowledge = "2024-12-31" open_weights = true [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 [limit] -context = 128_000 +context = 131_072 output = 96_000 [modalities] diff --git a/providers/openrouter/models/z-ai/glm-4.5.toml b/providers/openrouter/models/z-ai/glm-4.5.toml index 8530b1aa3..cad3d04ca 100644 --- a/providers/openrouter/models/z-ai/glm-4.5.toml +++ b/providers/openrouter/models/z-ai/glm-4.5.toml @@ -1,22 +1,23 @@ name = "GLM 4.5" family = "glm" -release_date = "2025-07-28" -last_updated = "2025-07-28" +release_date = "2025-07-25" +last_updated = "2025-07-25" attachment = false reasoning = true temperature = true tool_call = true structured_output = true -knowledge = "2025-04" +knowledge = "2024-12-31" open_weights = true [cost] -input = 0.60 -output = 2.20 +input = 0.6 +output = 2.2 +cache_read = 0.11 [limit] -context = 128_000 -output = 96_000 +context = 131_072 +output = 98_304 [modalities] input = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.5v.toml b/providers/openrouter/models/z-ai/glm-4.5v.toml index 344a52788..b37dc3f83 100644 --- a/providers/openrouter/models/z-ai/glm-4.5v.toml +++ b/providers/openrouter/models/z-ai/glm-4.5v.toml @@ -5,20 +5,20 @@ last_updated = "2025-08-11" attachment = true reasoning = true temperature = true -knowledge = "2025-04" tool_call = true -structured_output = true +structured_output = false +knowledge = "2024-12-31" open_weights = true [cost] input = 0.6 output = 1.8 +cache_read = 0.11 [limit] -context = 64_000 +context = 65_536 output = 16_384 - [modalities] -input = ["text", "image", "video"] +input = ["text", "image"] output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.6.toml b/providers/openrouter/models/z-ai/glm-4.6.toml index c07bbb89a..13e3ae971 100644 --- a/providers/openrouter/models/z-ai/glm-4.6.toml +++ b/providers/openrouter/models/z-ai/glm-4.6.toml @@ -7,17 +7,17 @@ reasoning = true temperature = true tool_call = true structured_output = true -knowledge = "2025-09" +knowledge = "2025-03-31" open_weights = true [cost] -input = 0.60 -output = 2.20 -cache_read = 0.11 +input = 0.43 +output = 1.74 +cache_read = 0.08 [limit] -context = 200_000 -output = 128_000 +context = 202_752 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.6:exacto.toml b/providers/openrouter/models/z-ai/glm-4.6:exacto.toml deleted file mode 100644 index 4f9b6d1d8..000000000 --- a/providers/openrouter/models/z-ai/glm-4.6:exacto.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "GLM 4.6 (exacto)" -family = "glm" -release_date = "2025-09-30" -last_updated = "2025-09-30" -attachment = false -reasoning = true -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-09" -open_weights = true - -[cost] -input = 0.60 -output = 1.90 -cache_read = 0.11 - -[limit] -context = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.6v.toml b/providers/openrouter/models/z-ai/glm-4.6v.toml new file mode 100644 index 000000000..32792bc3c --- /dev/null +++ b/providers/openrouter/models/z-ai/glm-4.6v.toml @@ -0,0 +1,23 @@ +name = "GLM 4.6V" +family = "glm" +release_date = "2025-12-08" +last_updated = "2025-12-08" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = true + +[cost] +input = 0.3 +output = 0.9 +cache_read = 0.05 + +[limit] +context = 131_072 +output = 24_000 + +[modalities] +input = ["image", "text", "video"] +output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.7-flash.toml b/providers/openrouter/models/z-ai/glm-4.7-flash.toml index 163a6da0c..e90fafc27 100644 --- a/providers/openrouter/models/z-ai/glm-4.7-flash.toml +++ b/providers/openrouter/models/z-ai/glm-4.7-flash.toml @@ -1,4 +1,4 @@ -name = "GLM-4.7-Flash" +name = "GLM 4.7 Flash" family = "glm" release_date = "2026-01-19" last_updated = "2026-01-19" @@ -12,15 +12,14 @@ open_weights = true [interleaved] field = "reasoning_details" - - [cost] -input = 0.07 +input = 0.06 output = 0.4 +cache_read = 0.01 [limit] -context = 200_000 -output = 65_535 +context = 202_752 +output = 16_384 [modalities] input = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-4.7.toml b/providers/openrouter/models/z-ai/glm-4.7.toml index 121318b57..fda664bad 100644 --- a/providers/openrouter/models/z-ai/glm-4.7.toml +++ b/providers/openrouter/models/z-ai/glm-4.7.toml @@ -1,4 +1,4 @@ -name = "GLM-4.7" +name = "GLM 4.7" family = "glm" release_date = "2025-12-22" last_updated = "2025-12-22" @@ -14,13 +14,13 @@ open_weights = true field = "reasoning_details" [cost] -input = 0.6 -output = 2.2 -cache_read = 0.11 +input = 0.4 +output = 1.75 +cache_read = 0.08 [limit] -context = 204800 -output = 131072 +context = 202_752 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-5-turbo.toml b/providers/openrouter/models/z-ai/glm-5-turbo.toml index f972bb16c..a261ee7d6 100644 --- a/providers/openrouter/models/z-ai/glm-5-turbo.toml +++ b/providers/openrouter/models/z-ai/glm-5-turbo.toml @@ -1,7 +1,7 @@ -name = "GLM-5-Turbo" +name = "GLM 5 Turbo" family = "glm" -release_date = "2026-03-16" -last_updated = "2026-03-16" +release_date = "2026-03-15" +last_updated = "2026-03-15" attachment = false reasoning = true temperature = true @@ -13,10 +13,9 @@ open_weights = false field = "reasoning_content" [cost] -input = 0.96 -output = 3.20 -cache_read = 0.192 -cache_write = 0 +input = 1.2 +output = 4 +cache_read = 0.24 [limit] context = 202_752 diff --git a/providers/openrouter/models/z-ai/glm-5.1.toml b/providers/openrouter/models/z-ai/glm-5.1.toml index 2a4713391..a4be499e2 100644 --- a/providers/openrouter/models/z-ai/glm-5.1.toml +++ b/providers/openrouter/models/z-ai/glm-5.1.toml @@ -1,4 +1,4 @@ -name = "GLM-5.1" +name = "GLM 5.1" family = "glm" release_date = "2026-04-07" last_updated = "2026-04-07" @@ -13,9 +13,9 @@ open_weights = true field = "reasoning_content" [cost] -input = 1.40 -output = 4.40 -cache_read = 0.26 +input = 0.98 +output = 3.08 +cache_read = 0.182 [limit] context = 202_752 diff --git a/providers/openrouter/models/z-ai/glm-5.toml b/providers/openrouter/models/z-ai/glm-5.toml index 6ee6a9ae1..2b51214d3 100644 --- a/providers/openrouter/models/z-ai/glm-5.toml +++ b/providers/openrouter/models/z-ai/glm-5.toml @@ -1,7 +1,7 @@ -name = "GLM-5" +name = "GLM 5" family = "glm" -release_date = "2026-02-12" -last_updated = "2026-02-12" +release_date = "2026-02-11" +last_updated = "2026-02-11" attachment = false reasoning = true temperature = true @@ -13,9 +13,9 @@ open_weights = true field = "reasoning_content" [cost] -input = 1.00 -output = 3.20 -cache_read = 0.2 +input = 0.6 +output = 1.92 +cache_read = 0.12 [limit] context = 202_752 diff --git a/providers/openrouter/models/z-ai/glm-5v-turbo.toml b/providers/openrouter/models/z-ai/glm-5v-turbo.toml new file mode 100644 index 000000000..003397582 --- /dev/null +++ b/providers/openrouter/models/z-ai/glm-5v-turbo.toml @@ -0,0 +1,23 @@ +name = "GLM 5V Turbo" +family = "glm" +release_date = "2026-04-01" +last_updated = "2026-04-01" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 1.2 +output = 4 +cache_read = 0.24 + +[limit] +context = 202_752 +output = 131_072 + +[modalities] +input = ["image", "text", "video"] +output = ["text"] diff --git a/providers/openrouter/models/~anthropic/claude-haiku-latest.toml b/providers/openrouter/models/~anthropic/claude-haiku-latest.toml new file mode 100644 index 000000000..d3f4691f0 --- /dev/null +++ b/providers/openrouter/models/~anthropic/claude-haiku-latest.toml @@ -0,0 +1,24 @@ +name = "Anthropic Claude Haiku Latest" +family = "claude-haiku" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 1 +output = 5 +cache_read = 0.1 +cache_write = 1.25 + +[limit] +context = 200_000 +output = 64_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/~anthropic/claude-opus-latest.toml b/providers/openrouter/models/~anthropic/claude-opus-latest.toml new file mode 100644 index 000000000..47ae423bc --- /dev/null +++ b/providers/openrouter/models/~anthropic/claude-opus-latest.toml @@ -0,0 +1,24 @@ +name = "Claude Opus Latest" +family = "claude-opus" +release_date = "2026-04-21" +last_updated = "2026-04-21" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-3.7-sonnet.toml b/providers/openrouter/models/~anthropic/claude-sonnet-latest.toml similarity index 52% rename from providers/openrouter/models/anthropic/claude-3.7-sonnet.toml rename to providers/openrouter/models/~anthropic/claude-sonnet-latest.toml index df3bf62c3..30bcc108a 100644 --- a/providers/openrouter/models/anthropic/claude-3.7-sonnet.toml +++ b/providers/openrouter/models/~anthropic/claude-sonnet-latest.toml @@ -1,22 +1,22 @@ -name = "Claude Sonnet 3.7" +name = "Anthropic Claude Sonnet Latest" family = "claude-sonnet" -release_date = "2025-02-19" -last_updated = "2025-02-19" +release_date = "2026-04-27" +last_updated = "2026-04-27" attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2024-01" +structured_output = true open_weights = false [cost] -input = 15.00 -output = 75.00 -cache_read = 1.50 -cache_write = 18.75 +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 [limit] -context = 200_000 +context = 1_000_000 output = 128_000 [modalities] diff --git a/providers/openrouter/models/google/gemini-2.5-flash-preview-09-2025.toml b/providers/openrouter/models/~google/gemini-flash-latest.toml similarity index 50% rename from providers/openrouter/models/google/gemini-2.5-flash-preview-09-2025.toml rename to providers/openrouter/models/~google/gemini-flash-latest.toml index fc12bfd01..ca2c88cc5 100644 --- a/providers/openrouter/models/google/gemini-2.5-flash-preview-09-2025.toml +++ b/providers/openrouter/models/~google/gemini-flash-latest.toml @@ -1,24 +1,25 @@ -name = "Gemini 2.5 Flash Preview 09-25" +name = "Google Gemini Flash Latest" family = "gemini-flash" -release_date = "2025-09-25" -last_updated = "2025-09-25" +release_date = "2026-04-27" +last_updated = "2026-04-27" attachment = true reasoning = true temperature = true -knowledge = "2025-01" tool_call = true structured_output = true open_weights = false [cost] -input = 0.30 -output = 2.50 -cache_read = 0.031 +input = 0.5 +output = 3 +reasoning = 3 +cache_read = 0.05 +cache_write = 0.083333 [limit] context = 1_048_576 output = 65_536 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "pdf", "audio", "video"] output = ["text"] diff --git a/providers/openrouter/models/~google/gemini-pro-latest.toml b/providers/openrouter/models/~google/gemini-pro-latest.toml new file mode 100644 index 000000000..2ac373dd8 --- /dev/null +++ b/providers/openrouter/models/~google/gemini-pro-latest.toml @@ -0,0 +1,25 @@ +name = "Google Gemini Pro Latest" +family = "gemini-pro" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 2 +output = 12 +reasoning = 12 +cache_read = 0.2 +cache_write = 0.375 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["audio", "pdf", "image", "text", "video"] +output = ["text"] diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml new file mode 100644 index 000000000..03f6d2181 --- /dev/null +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -0,0 +1,23 @@ +name = "MoonshotAI Kimi Latest" +family = "kimi" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[cost] +input = 0.73 +output = 3.49 +cache_read = 0.25 + +[limit] +context = 262_142 +output = 262_142 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/~openai/gpt-latest.toml b/providers/openrouter/models/~openai/gpt-latest.toml new file mode 100644 index 000000000..338040f6f --- /dev/null +++ b/providers/openrouter/models/~openai/gpt-latest.toml @@ -0,0 +1,24 @@ +name = "OpenAI GPT Latest" +family = "gpt" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2025-12-01" +open_weights = false + +[cost] +input = 5 +output = 30 +cache_read = 0.5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/~openai/gpt-mini-latest.toml b/providers/openrouter/models/~openai/gpt-mini-latest.toml new file mode 100644 index 000000000..e874a1aa9 --- /dev/null +++ b/providers/openrouter/models/~openai/gpt-mini-latest.toml @@ -0,0 +1,24 @@ +name = "OpenAI GPT Mini Latest" +family = "gpt-mini" +release_date = "2026-04-27" +last_updated = "2026-04-27" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2025-08-31" +open_weights = false + +[cost] +input = 0.75 +output = 4.5 +cache_read = 0.075 + +[limit] +context = 400_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"]