From 4e85eac00a6b2d19c36cd6d86cd217d28ed0f917 Mon Sep 17 00:00:00 2001 From: Aiden Cline Date: Fri, 26 Jun 2026 09:37:21 -0500 Subject: [PATCH] docs: document provider reasoning request formats --- providers/302ai/provider.toml | 6 +++++ providers/abacus/provider.toml | 6 +++++ .../models/abliterated-model.toml | 6 +++++ providers/abliteration-ai/provider.toml | 7 ++++++ .../aihubmix/models/claude-opus-4-6.toml | 1 + .../aihubmix/models/claude-opus-4-7.toml | 1 + .../aihubmix/models/claude-sonnet-4-6.toml | 1 + .../aihubmix/models/gemini-2.5-flash.toml | 1 + providers/aihubmix/models/gemini-2.5-pro.toml | 1 + providers/aihubmix/provider.toml | 4 ++++ providers/alibaba-cn/provider.toml | 15 +++++++++++++ .../models/MiniMax-M2.5.toml | 6 +++++ .../alibaba-coding-plan-cn/models/glm-5.toml | 6 +++++ .../models/kimi-k2.5.toml | 7 ++++++ .../models/qwen3-coder-plus.toml | 5 +++++ .../models/qwen3.7-plus.toml | 7 ++++++ .../alibaba-coding-plan-cn/provider.toml | 10 +++++++++ .../models/MiniMax-M2.5.toml | 6 +++++ .../alibaba-coding-plan/models/glm-5.toml | 6 +++++ .../alibaba-coding-plan/models/kimi-k2.5.toml | 7 ++++++ .../models/qwen3-coder-plus.toml | 5 +++++ .../models/qwen3.7-plus.toml | 7 ++++++ providers/alibaba-coding-plan/provider.toml | 10 +++++++++ .../models/MiniMax-M2.5.toml | 5 +++++ .../models/deepseek-v4-flash.toml | 7 ++++++ .../models/deepseek-v4-pro.toml | 7 ++++++ .../models/kimi-k2.5.toml | 6 +++++ .../models/kimi-k2.6.toml | 6 +++++ .../models/kimi-k2.7-code.toml | 6 +++++ .../models/qwen3.6-flash.toml | 7 ++++++ .../models/qwen3.6-plus.toml | 7 ++++++ .../models/qwen3.7-max.toml | 7 ++++++ .../models/qwen3.7-plus.toml | 7 ++++++ providers/alibaba-token-plan-cn/provider.toml | 12 ++++++++++ .../models/MiniMax-M2.5.toml | 5 +++++ .../models/deepseek-v4-flash.toml | 7 ++++++ .../models/deepseek-v4-pro.toml | 7 ++++++ .../alibaba-token-plan/models/kimi-k2.5.toml | 6 +++++ .../alibaba-token-plan/models/kimi-k2.6.toml | 6 +++++ .../models/kimi-k2.7-code.toml | 6 +++++ .../models/qwen3.6-flash.toml | 7 ++++++ .../models/qwen3.6-plus.toml | 7 ++++++ .../models/qwen3.7-max.toml | 7 ++++++ .../models/qwen3.7-plus.toml | 7 ++++++ providers/alibaba-token-plan/provider.toml | 12 ++++++++++ providers/alibaba/provider.toml | 15 +++++++++++++ .../ambient/models/moonshotai/kimi-k2.6.toml | 2 ++ .../ambient/models/zai-org/GLM-5.1-FP8.toml | 2 ++ providers/ambient/provider.toml | 6 +++++ .../models/anthropic/claude-haiku-4-5.toml | 4 ++++ .../models/anthropic/claude-opus-4-6.toml | 4 ++++ .../models/anthropic/claude-opus-4-7.toml | 3 +++ .../models/anthropic/claude-sonnet-4-5.toml | 4 ++++ .../models/anthropic/claude-sonnet-4-6.toml | 4 ++++ .../anyapi/models/openai/gpt-5-mini.toml | 3 +++ providers/anyapi/models/openai/gpt-5.1.toml | 3 +++ providers/anyapi/models/openai/gpt-5.2.toml | 3 +++ providers/anyapi/models/openai/gpt-5.4.toml | 3 +++ providers/anyapi/models/openai/gpt-5.toml | 3 +++ providers/anyapi/models/openai/o3-mini.toml | 3 +++ providers/anyapi/models/openai/o3.toml | 3 +++ providers/anyapi/models/openai/o4-mini.toml | 3 +++ providers/anyapi/provider.toml | 8 +++++++ providers/atomic-chat/provider.toml | 8 +++++++ providers/auriko/models/claude-opus-4-6.toml | 3 +++ providers/auriko/models/claude-opus-4-7.toml | 3 +++ .../auriko/models/claude-sonnet-4-6.toml | 3 +++ .../auriko/models/deepseek-v4-flash.toml | 3 +++ providers/auriko/models/deepseek-v4-pro.toml | 2 ++ providers/auriko/models/gemini-2.5-flash.toml | 3 +++ providers/auriko/models/gemini-2.5-pro.toml | 3 +++ .../auriko/models/gemini-3.1-pro-preview.toml | 3 +++ providers/auriko/models/glm-5.1.toml | 3 +++ providers/auriko/models/grok-4.3.toml | 3 +++ providers/auriko/models/kimi-k2.5.toml | 2 ++ providers/auriko/models/kimi-k2.6.toml | 2 ++ .../auriko/models/minimax-m2-7-highspeed.toml | 3 +++ providers/auriko/models/minimax-m2-7.toml | 3 +++ providers/auriko/models/qwen-3.6-plus.toml | 3 +++ providers/auriko/provider.toml | 9 ++++++++ .../models/claude-haiku-4-5.toml | 3 +++ .../models/claude-opus-4-1.toml | 3 +++ .../models/claude-opus-4-5.toml | 3 +++ .../models/claude-opus-4-6.toml | 3 +++ .../models/claude-opus-4-8.toml | 3 +++ .../models/claude-sonnet-4-5.toml | 3 +++ .../models/gpt-5.4-mini.toml | 3 +++ .../models/gpt-5.4-nano.toml | 3 +++ .../models/gpt-5.4-pro.toml | 3 +++ .../models/gpt-5.4.toml | 3 +++ .../models/kimi-k2.5.toml | 3 +++ .../models/kimi-k2.6.toml | 3 +++ .../azure-cognitive-services/provider.toml | 8 +++++++ providers/azure/models/deepseek-v4-flash.toml | 3 +++ providers/azure/models/deepseek-v4-pro.toml | 3 +++ providers/azure/models/kimi-k2.5.toml | 3 +++ providers/azure/models/kimi-k2.6.toml | 3 +++ providers/bailing/models/Ring-1T.toml | 2 ++ providers/bailing/provider.toml | 5 +++++ .../models/MiniMaxAI/MiniMax-M2.5.toml | 3 +++ .../models/deepseek-ai/DeepSeek-V3.1.toml | 3 +++ .../models/deepseek-ai/DeepSeek-V4-Pro.toml | 3 +++ .../baseten/models/moonshotai/Kimi-K2.5.toml | 3 +++ .../baseten/models/moonshotai/Kimi-K2.6.toml | 3 +++ .../models/moonshotai/Kimi-K2.7-Code.toml | 3 +++ .../NVIDIA-Nemotron-3-Ultra-550B-A55B.toml | 3 +++ .../models/nvidia/Nemotron-120B-A12B.toml | 3 +++ .../baseten/models/openai/gpt-oss-120b.toml | 3 +++ providers/baseten/models/zai-org/GLM-4.7.toml | 3 +++ providers/baseten/models/zai-org/GLM-5.1.toml | 3 +++ providers/baseten/models/zai-org/GLM-5.2.toml | 3 +++ providers/baseten/models/zai-org/GLM-5.toml | 3 +++ providers/baseten/provider.toml | 9 ++++++++ .../berget/models/google/gemma-4-31B-it.toml | 2 ++ .../meta-llama/Llama-3.3-70B-Instruct.toml | 2 ++ .../mistralai/Mistral-Medium-3.5-128B.toml | 3 +++ .../Mistral-Small-3.2-24B-Instruct-2506.toml | 2 ++ .../berget/models/moonshotai/Kimi-K2.6.toml | 3 +++ .../berget/models/openai/gpt-oss-120b.toml | 3 +++ providers/berget/models/zai-org/GLM-4.7.toml | 2 ++ providers/berget/models/zai-org/GLM-5.2.toml | 3 +++ providers/berget/provider.toml | 7 ++++++ providers/cerebras/models/gpt-oss-120b.toml | 5 +++++ providers/cerebras/models/zai-glm-4.7.toml | 7 ++++++ providers/cerebras/provider.toml | 8 +++++++ providers/chutes/provider.toml | 7 ++++++ .../models/gpt-oss-120b-high-throughput.toml | 3 +++ .../chat-completion/models/gpt-oss-20b.toml | 3 +++ providers/clarifai/provider.toml | 6 +++++ providers/claudinio/models/claudinio.toml | 3 +++ providers/claudinio/provider.toml | 4 ++++ .../models/MiniMaxAI/MiniMax-M2.5.toml | 3 +++ .../models/openai/gpt-oss-120b.toml | 3 +++ providers/cloudferro-sherlock/provider.toml | 4 ++++ .../workers-ai/@cf/moonshotai/kimi-k2.5.toml | 3 +++ .../workers-ai/@cf/moonshotai/kimi-k2.6.toml | 3 +++ .../@cf/nvidia/nemotron-3-120b-a12b.toml | 3 +++ .../workers-ai/@cf/zai-org/glm-4.7-flash.toml | 3 +++ providers/cloudflare-ai-gateway/provider.toml | 11 ++++++++++ .../deepseek-r1-distill-qwen-32b.toml | 3 +++ .../models/@cf/google/gemma-4-26b-a4b-it.toml | 3 +++ .../models/@cf/moonshotai/kimi-k2.6.toml | 3 +++ .../@cf/nvidia/nemotron-3-120b-a12b.toml | 3 +++ .../models/@cf/openai/gpt-oss-120b.toml | 3 +++ .../models/@cf/openai/gpt-oss-20b.toml | 3 +++ .../models/@cf/qwen/qwen3-30b-a3b-fp8.toml | 3 +++ .../models/@cf/qwen/qwq-32b.toml | 3 +++ .../models/@cf/zai-org/glm-4.7-flash.toml | 3 +++ providers/cloudflare-workers-ai/provider.toml | 8 +++++++ .../cohere/models/command-a-plus-05-2026.toml | 8 +++++++ .../models/command-a-reasoning-08-2025.toml | 8 +++++++ .../cohere/models/north-mini-code-1-0.toml | 5 +++++ providers/cohere/provider.toml | 11 ++++++++++ .../cortecs/models/claude-4-5-sonnet.toml | 3 +++ .../cortecs/models/claude-4-6-sonnet.toml | 3 +++ .../cortecs/models/claude-haiku-4-5.toml | 3 +++ providers/cortecs/models/claude-opus4-5.toml | 3 +++ providers/cortecs/models/claude-opus4-6.toml | 3 +++ providers/cortecs/models/claude-opus4-7.toml | 3 +++ providers/cortecs/models/claude-opus4-8.toml | 3 +++ .../cortecs/models/deepseek-v4-flash.toml | 2 ++ providers/cortecs/models/deepseek-v4-pro.toml | 2 ++ providers/cortecs/models/glm-5.2.toml | 2 ++ providers/cortecs/models/gpt-5.4.toml | 2 ++ providers/cortecs/models/gpt-oss-120b.toml | 2 ++ providers/cortecs/models/kimi-k2.5.toml | 2 ++ providers/cortecs/models/kimi-k2.6.toml | 2 ++ providers/cortecs/provider.toml | 6 +++++ providers/crof/models/glm-4.7-flash.toml | 2 ++ providers/crof/models/glm-4.7.toml | 2 ++ providers/crof/models/glm-5.toml | 2 ++ providers/crof/models/minimax-m2.5.toml | 2 ++ providers/crof/provider.toml | 5 +++++ .../models/databricks-claude-haiku-4-5.toml | 3 +++ .../models/databricks-claude-opus-4-1.toml | 3 +++ .../models/databricks-claude-opus-4-5.toml | 3 +++ .../models/databricks-claude-opus-4-6.toml | 3 +++ .../models/databricks-claude-opus-4-7.toml | 3 +++ .../models/databricks-claude-sonnet-4-5.toml | 3 +++ .../models/databricks-claude-sonnet-4-6.toml | 3 +++ .../models/databricks-claude-sonnet-4.toml | 3 +++ .../models/databricks-gemini-2-5-flash.toml | 3 +++ .../models/databricks-gemini-2-5-pro.toml | 3 +++ .../databricks-gemini-3-1-flash-lite.toml | 2 ++ .../models/databricks-gemini-3-1-pro.toml | 2 ++ .../models/databricks-gemini-3-flash.toml | 2 ++ .../models/databricks-gemini-3-pro.toml | 2 ++ .../databricks/models/databricks-gpt-5-1.toml | 2 ++ .../databricks/models/databricks-gpt-5-2.toml | 2 ++ .../models/databricks-gpt-5-4-mini.toml | 2 ++ .../models/databricks-gpt-5-4-nano.toml | 2 ++ .../databricks/models/databricks-gpt-5-4.toml | 2 ++ .../databricks/models/databricks-gpt-5-5.toml | 2 ++ .../models/databricks-gpt-5-mini.toml | 2 ++ .../models/databricks-gpt-5-nano.toml | 2 ++ .../databricks/models/databricks-gpt-5.toml | 2 ++ .../models/databricks-gpt-oss-120b.toml | 2 ++ .../models/databricks-gpt-oss-20b.toml | 2 ++ providers/databricks/provider.toml | 10 +++++++++ .../models/deepseek-ai/DeepSeek-V4-Flash.toml | 5 +++++ .../models/deepseek-ai/DeepSeek-V4-Pro.toml | 5 +++++ .../deepinfra/models/openai/gpt-oss-120b.toml | 5 +++++ .../deepinfra/models/openai/gpt-oss-20b.toml | 5 +++++ providers/deepinfra/provider.toml | 14 ++++++++++++ .../deepseek/models/deepseek-reasoner.toml | 3 +++ .../deepseek/models/deepseek-v4-flash.toml | 3 +++ .../deepseek/models/deepseek-v4-pro.toml | 3 +++ providers/deepseek/provider.toml | 6 +++++ .../models/openai-gpt-5.4-mini.toml | 4 ++++ .../models/openai-gpt-5.4-nano.toml | 4 ++++ .../models/openai-gpt-5.4-pro.toml | 4 ++++ .../digitalocean/models/openai-gpt-5.4.toml | 4 ++++ .../digitalocean/models/openai-gpt-5.5.toml | 4 ++++ providers/digitalocean/provider.toml | 7 ++++++ providers/dinference/provider.toml | 3 +++ providers/drun/provider.toml | 3 +++ providers/evroc/provider.toml | 4 ++++ providers/fastrouter/provider.toml | 5 +++++ .../fireworks/models/deepseek-v4-flash.toml | 2 ++ .../fireworks/models/deepseek-v4-pro.toml | 2 ++ .../accounts/fireworks/models/glm-5p1.toml | 2 ++ .../accounts/fireworks/models/glm-5p2.toml | 2 ++ .../fireworks/models/gpt-oss-120b.toml | 2 ++ .../fireworks/models/gpt-oss-20b.toml | 2 ++ .../accounts/fireworks/models/kimi-k2p6.toml | 2 ++ .../fireworks/models/kimi-k2p7-code.toml | 2 ++ .../fireworks/models/minimax-m2p7.toml | 2 ++ .../accounts/fireworks/models/minimax-m3.toml | 2 ++ .../fireworks/models/qwen3p7-plus.toml | 3 +++ .../fireworks/routers/glm-5p1-fast.toml | 2 ++ .../fireworks/routers/kimi-k2p6-fast.toml | 2 ++ .../fireworks/routers/kimi-k2p6-turbo.toml | 2 ++ .../routers/kimi-k2p7-code-fast.toml | 2 ++ providers/fireworks-ai/provider.toml | 4 ++++ providers/freemodel/provider.toml | 4 ++++ .../friendli/models/zai-org/GLM-5.2.toml | 3 +++ providers/friendli/provider.toml | 4 ++++ providers/frogbot/provider.toml | 7 ++++++ providers/github-copilot/provider.toml | 6 +++++ providers/github-models/provider.toml | 4 ++++ providers/gitlab/provider.toml | 17 ++++++++++++++ providers/gmicloud/provider.toml | 4 ++++ .../deepseek-ai/deepseek-v3.1-maas.toml | 3 +++ .../deepseek-ai/deepseek-v3.2-maas.toml | 3 +++ .../meta/llama-3.3-70b-instruct-maas.toml | 3 +++ ...ama-4-maverick-17b-128e-instruct-maas.toml | 3 +++ .../moonshotai/kimi-k2-thinking-maas.toml | 3 +++ .../qwen3-235b-a22b-instruct-2507-maas.toml | 3 +++ .../models/zai-org/glm-4.7-maas.toml | 3 +++ .../models/zai-org/glm-5-maas.toml | 3 +++ .../groq/models/openai/gpt-oss-120b.toml | 4 ++++ providers/groq/models/openai/gpt-oss-20b.toml | 4 ++++ .../models/openai/gpt-oss-safeguard-20b.toml | 8 +++++++ providers/groq/models/qwen/qwen3-32b.toml | 8 +++++++ providers/groq/provider.toml | 7 ++++++ .../helicone/models/claude-4.5-opus.toml | 3 +++ .../helicone/models/claude-4.5-sonnet.toml | 3 +++ .../models/claude-opus-4-1-20250805.toml | 3 +++ .../helicone/models/claude-opus-4-1.toml | 3 +++ providers/helicone/models/claude-opus-4.toml | 3 +++ .../models/claude-sonnet-4-5-20250929.toml | 3 +++ .../helicone/models/claude-sonnet-4.toml | 3 +++ .../models/gemini-2.5-flash-lite.toml | 3 +++ .../helicone/models/gemini-2.5-flash.toml | 3 +++ providers/helicone/models/gemini-2.5-pro.toml | 3 +++ .../helicone/models/gemini-3-pro-preview.toml | 3 +++ providers/helicone/models/gpt-oss-120b.toml | 3 +++ providers/helicone/models/gpt-oss-20b.toml | 3 +++ .../helicone/models/sonar-deep-research.toml | 3 +++ .../helicone/models/sonar-reasoning-pro.toml | 3 +++ providers/helicone/provider.toml | 22 +++++++++++++++++++ providers/hpc-ai/provider.toml | 4 ++++ providers/huggingface/provider.toml | 6 +++++ providers/iflowcn/provider.toml | 7 ++++++ providers/inception/models/mercury-2.toml | 3 +++ .../inception/models/mercury-edit-2.toml | 3 +++ providers/inception/provider.toml | 6 +++++ providers/inceptron/provider.toml | 4 ++++ providers/inference/provider.toml | 4 ++++ providers/io-net/provider.toml | 6 +++++ providers/jiekou/provider.toml | 16 ++++++++++++++ providers/kilo/provider.toml | 19 ++++++++++++++++ .../models/GLM-4.7.toml | 3 +++ .../kuae-cloud-coding-plan/provider.toml | 3 +++ .../lilac/models/google/gemma-4-31b-it.toml | 3 +++ .../lilac/models/minimaxai/minimax-m3.toml | 3 +++ .../lilac/models/moonshotai/kimi-k2.6.toml | 3 +++ providers/lilac/models/zai-org/glm-5.2.toml | 4 ++++ providers/lilac/provider.toml | 3 +++ providers/llama/provider.toml | 3 +++ .../models/grok-4-1-fast-reasoning.toml | 3 +++ .../models/grok-4-20-beta-0309-reasoning.toml | 3 +++ .../models/grok-4-20-reasoning.toml | 3 +++ providers/llmgateway/models/grok-4-3.toml | 3 +++ .../llmgateway/models/grok-build-0-1.toml | 3 +++ .../models/kimi-k2.7-code-highspeed.toml | 3 +++ .../llmgateway/models/kimi-k2.7-code.toml | 3 +++ providers/llmgateway/models/mimo-v2-pro.toml | 3 +++ .../llmgateway/models/mimo-v2.5-pro.toml | 3 +++ providers/llmgateway/models/mimo-v2.5.toml | 3 +++ .../models/nemotron-3-ultra-550b.toml | 3 +++ .../models/sonar-reasoning-pro.toml | 3 +++ providers/llmgateway/provider.toml | 8 +++++++ providers/llmtr/models/qwen3-6-35b.toml | 3 +++ providers/llmtr/provider.toml | 9 ++++++++ .../lmstudio/models/openai/gpt-oss-20b.toml | 3 +++ providers/lmstudio/provider.toml | 6 +++++ providers/lucidquery/provider.toml | 4 ++++ providers/meganova/models/zai-org/GLM-5.toml | 3 +++ providers/meganova/provider.toml | 3 +++ providers/merge-gateway/provider.toml | 16 ++++++++++++++ .../models/qwen/qwen3.5-122b-a10b.toml | 2 ++ .../mixlayer/models/qwen/qwen3.5-27b.toml | 2 ++ .../mixlayer/models/qwen/qwen3.5-35b-a3b.toml | 2 ++ .../models/qwen/qwen3.5-397b-a17b.toml | 2 ++ .../mixlayer/models/qwen/qwen3.5-9b.toml | 2 ++ providers/mixlayer/provider.toml | 4 ++++ providers/moark/models/GLM-4.7.toml | 3 +++ providers/moark/models/MiniMax-M2.1.toml | 3 +++ providers/moark/provider.toml | 3 +++ providers/modelscope/provider.toml | 3 +++ providers/moonshotai-cn/provider.toml | 4 ++++ providers/moonshotai/provider.toml | 4 ++++ providers/morph/provider.toml | 4 ++++ providers/nano-gpt/provider.toml | 10 +++++++++ providers/nearai/provider.toml | 6 +++++ providers/nebius/provider.toml | 5 +++++ providers/neon/provider.toml | 10 +++++++++ providers/neuralwatt/provider.toml | 5 +++++ providers/nova/provider.toml | 6 +++++ providers/novita-ai/provider.toml | 6 +++++ .../models/deepseek-ai/deepseek-v4-flash.toml | 2 ++ .../models/deepseek-ai/deepseek-v4-pro.toml | 2 ++ ...emotron-3-nano-omni-30b-a3b-reasoning.toml | 4 ++++ .../nvidia/models/qwen/qwen3.5-122b-a10b.toml | 2 ++ .../nvidia/models/qwen/qwen3.5-397b-a17b.toml | 2 ++ providers/nvidia/provider.toml | 6 +++++ providers/ollama-cloud/provider.toml | 8 +++++++ providers/opencode-go/provider.toml | 6 +++++ providers/opencode/provider.toml | 7 ++++++ .../anthropic/claude-opus-4.6-fast.toml | 3 ++- .../anthropic/claude-opus-4.7-fast.toml | 3 ++- .../anthropic/claude-opus-4.8-fast.toml | 3 ++- .../~anthropic/claude-fable-latest.toml | 3 ++- .../models/~anthropic/claude-opus-latest.toml | 3 ++- .../~anthropic/claude-sonnet-latest.toml | 3 ++- providers/openrouter/provider.toml | 12 ++++++++++ providers/orcarouter/provider.toml | 11 ++++++++++ providers/ovhcloud/models/gpt-oss-120b.toml | 2 ++ providers/ovhcloud/models/gpt-oss-20b.toml | 2 ++ providers/ovhcloud/models/qwen3-32b.toml | 2 ++ .../ovhcloud/models/qwen3.5-397b-a17b.toml | 2 ++ providers/ovhcloud/models/qwen3.5-9b.toml | 2 ++ providers/ovhcloud/models/qwen3.6-27b.toml | 2 ++ providers/ovhcloud/provider.toml | 6 +++++ .../models/sonar-deep-research.toml | 7 ++++++ .../models/sonar-reasoning-pro.toml | 7 ++++++ providers/perplexity/provider.toml | 6 +++++ providers/poe/provider.toml | 7 ++++++ .../poolside/models/poolside/laguna-m.1.toml | 3 +++ .../poolside/models/poolside/laguna-xs.2.toml | 3 +++ providers/poolside/provider.toml | 8 +++++++ providers/privatemode-ai/provider.toml | 6 +++++ providers/qihang-ai/provider.toml | 5 +++++ providers/qiniu-ai/models/mimo-v2-flash.toml | 3 +++ .../qiniu-ai/models/xiaomi/mimo-v2-flash.toml | 3 +++ providers/qiniu-ai/provider.toml | 7 ++++++ providers/regolo-ai/models/gpt-oss-120b.toml | 3 +++ providers/regolo-ai/models/gpt-oss-20b.toml | 3 +++ providers/regolo-ai/provider.toml | 6 +++++ providers/requesty/provider.toml | 10 +++++++++ providers/routing-run/provider.toml | 5 +++++ .../models/anthropic--claude-3.7-sonnet.toml | 1 + .../models/anthropic--claude-4.6-opus.toml | 1 + .../models/anthropic--claude-4.6-sonnet.toml | 1 + .../models/anthropic--claude-4.7-opus.toml | 1 + .../models/gemini-2.5-flash-lite.toml | 1 + .../sap-ai-core/models/gemini-2.5-flash.toml | 1 + .../sap-ai-core/models/gemini-2.5-pro.toml | 1 + providers/sap-ai-core/models/gpt-5.4.toml | 1 + providers/sap-ai-core/models/gpt-5.5.toml | 1 + providers/sap-ai-core/models/gpt-5.toml | 1 + providers/sap-ai-core/provider.toml | 1 + providers/sarvam/provider.toml | 7 ++++++ providers/scaleway/models/gemma-3-27b-it.toml | 3 +++ providers/scaleway/models/gpt-oss-120b.toml | 3 +++ .../models/qwen3-235b-a22b-instruct-2507.toml | 3 +++ providers/scaleway/provider.toml | 7 ++++++ .../models/deepseek-ai/DeepSeek-V4-Flash.toml | 4 ++++ .../models/tencent/Hunyuan-A13B-Instruct.toml | 3 +++ providers/siliconflow-cn/provider.toml | 8 +++++++ .../models/tencent/Hunyuan-A13B-Instruct.toml | 2 ++ providers/siliconflow/provider.toml | 7 ++++++ providers/snowflake-cortex/provider.toml | 11 ++++++++++ providers/stackit/provider.toml | 5 +++++ providers/stepfun-ai/provider.toml | 7 ++++++ .../stepfun/models/step-3.5-flash-2603.toml | 3 +++ providers/stepfun/models/step-3.7-flash.toml | 3 +++ providers/stepfun/provider.toml | 12 ++++++++++ providers/submodel/provider.toml | 7 ++++++ providers/synthetic/provider.toml | 8 +++++++ .../tencent-coding-plan/models/glm-5.toml | 3 +++ .../tencent-coding-plan/models/kimi-k2.5.toml | 3 +++ providers/tencent-coding-plan/provider.toml | 9 ++++++++ .../tencent-tokenhub/models/hy3-preview.toml | 3 +++ providers/tencent-tokenhub/provider.toml | 8 +++++++ providers/the-grid-ai/provider.toml | 10 +++++++++ .../models/MiniMaxAI/MiniMax-M2.5.toml | 4 ++++ .../models/MiniMaxAI/MiniMax-M2.7.toml | 4 ++++ .../models/Qwen/Qwen3.5-397B-A17B.toml | 5 +++++ .../togetherai/models/Qwen/Qwen3.5-9B.toml | 4 ++++ .../togetherai/models/Qwen/Qwen3.6-Plus.toml | 4 ++++ .../models/deepcogito/cogito-v2-1-671b.toml | 4 ++++ .../models/deepseek-ai/DeepSeek-R1.toml | 4 ++++ .../models/deepseek-ai/DeepSeek-V3-1.toml | 4 ++++ .../models/deepseek-ai/DeepSeek-V4-Pro.toml | 5 +++++ .../models/google/gemma-4-31B-it.toml | 4 ++++ .../models/moonshotai/Kimi-K2.5.toml | 5 +++++ .../models/moonshotai/Kimi-K2.6.toml | 4 ++++ .../nvidia/nemotron-3-ultra-550b-a55b.toml | 5 +++++ .../models/openai/gpt-oss-120b.toml | 4 ++++ .../togetherai/models/openai/gpt-oss-20b.toml | 4 ++++ .../models/pearl-ai/gemma-4-31b-it.toml | 4 ++++ .../togetherai/models/zai-org/GLM-5.1.toml | 4 ++++ .../togetherai/models/zai-org/GLM-5.toml | 4 ++++ providers/togetherai/provider.toml | 9 ++++++++ .../models/umans-coder.toml | 3 +++ .../models/umans-flash.toml | 3 +++ .../models/umans-glm-5.1.toml | 2 ++ .../models/umans-glm-5.2.toml | 2 ++ .../models/umans-kimi-k2.7.toml | 2 ++ .../models/umans-qwen3.6-35b-a3b.toml | 3 +++ providers/umans-ai-coding-plan/provider.toml | 11 ++++++++++ providers/umans-ai/models/umans-coder.toml | 3 +++ providers/umans-ai/models/umans-flash.toml | 3 +++ providers/umans-ai/models/umans-glm-5.1.toml | 2 ++ providers/umans-ai/models/umans-glm-5.2.toml | 2 ++ .../umans-ai/models/umans-kimi-k2.7.toml | 2 ++ providers/umans-ai/provider.toml | 11 ++++++++++ providers/upstage/provider.toml | 6 +++++ providers/v0/provider.toml | 2 ++ .../venice/models/aion-labs-aion-2-0.toml | 2 ++ .../models/arcee-trinity-large-thinking.toml | 2 ++ providers/venice/models/claude-fable-5.toml | 2 ++ providers/venice/models/claude-opus-4-5.toml | 2 ++ .../venice/models/claude-opus-4-6-fast.toml | 2 ++ providers/venice/models/claude-opus-4-6.toml | 2 ++ .../venice/models/claude-opus-4-7-fast.toml | 2 ++ providers/venice/models/claude-opus-4-7.toml | 2 ++ .../venice/models/claude-opus-4-8-fast.toml | 2 ++ providers/venice/models/claude-opus-4-8.toml | 2 ++ .../venice/models/claude-sonnet-4-5.toml | 2 ++ .../venice/models/claude-sonnet-4-6.toml | 2 ++ providers/venice/models/deepseek-v3.2.toml | 2 ++ .../venice/models/deepseek-v4-flash.toml | 2 ++ providers/venice/models/deepseek-v4-pro.toml | 2 ++ .../venice/models/gemini-3-1-pro-preview.toml | 2 ++ providers/venice/models/gemini-3-5-flash.toml | 2 ++ .../venice/models/gemini-3-flash-preview.toml | 2 ++ .../models/google-gemma-4-26b-a4b-it.toml | 2 ++ .../venice/models/google-gemma-4-31b-it.toml | 2 ++ .../venice/models/grok-4-20-multi-agent.toml | 2 ++ providers/venice/models/grok-4-20.toml | 2 ++ providers/venice/models/grok-4-3.toml | 2 ++ providers/venice/models/grok-build-0-1.toml | 2 ++ providers/venice/models/kimi-k2-5.toml | 2 ++ providers/venice/models/kimi-k2-6.toml | 2 ++ providers/venice/models/kimi-k2-7-code.toml | 2 ++ providers/venice/models/mercury-2.toml | 2 ++ providers/venice/models/minimax-m25.toml | 2 ++ providers/venice/models/minimax-m27.toml | 2 ++ .../venice/models/minimax-m3-preview.toml | 2 ++ .../venice/models/mistral-small-2603.toml | 2 ++ .../nvidia-nemotron-3-ultra-550b-a55b.toml | 2 ++ .../nvidia-nemotron-cascade-2-30b-a3b.toml | 2 ++ .../olafangensan-glm-4.7-flash-heretic.toml | 2 ++ .../venice/models/openai-gpt-52-codex.toml | 2 ++ providers/venice/models/openai-gpt-52.toml | 2 ++ .../venice/models/openai-gpt-53-codex.toml | 2 ++ .../venice/models/openai-gpt-54-mini.toml | 2 ++ .../venice/models/openai-gpt-54-pro.toml | 2 ++ providers/venice/models/openai-gpt-54.toml | 2 ++ .../venice/models/openai-gpt-55-pro.toml | 2 ++ providers/venice/models/openai-gpt-55.toml | 2 ++ .../venice/models/openai-gpt-oss-120b.toml | 2 ++ providers/venice/models/qwen-3-6-plus.toml | 2 ++ providers/venice/models/qwen-3-7-max.toml | 2 ++ providers/venice/models/qwen-3-7-plus.toml | 2 ++ .../models/qwen3-235b-a22b-thinking-2507.toml | 2 ++ providers/venice/models/qwen3-5-35b-a3b.toml | 2 ++ .../venice/models/qwen3-5-397b-a17b.toml | 2 ++ providers/venice/models/qwen3-5-9b.toml | 2 ++ providers/venice/models/qwen3-6-27b.toml | 2 ++ providers/venice/models/xiaomi-mimo-v2-5.toml | 2 ++ providers/venice/models/z-ai-glm-5-turbo.toml | 2 ++ .../venice/models/z-ai-glm-5v-turbo.toml | 2 ++ providers/venice/models/zai-org-glm-4.6.toml | 2 ++ .../venice/models/zai-org-glm-4.7-flash.toml | 2 ++ providers/venice/models/zai-org-glm-4.7.toml | 2 ++ providers/venice/models/zai-org-glm-5-1.toml | 2 ++ providers/venice/models/zai-org-glm-5-2.toml | 2 ++ providers/venice/models/zai-org-glm-5.toml | 2 ++ providers/venice/provider.toml | 17 ++++++++++++++ .../vercel/models/alibaba/qwen3.7-plus.toml | 9 ++++++++ providers/vercel/provider.toml | 10 +++++++++ providers/vivgrid/models/deepseek-v3.2.toml | 4 ++++ providers/vivgrid/models/deepseek-v4-pro.toml | 4 ++++ .../models/gemini-3.1-flash-lite-preview.toml | 4 ++++ .../models/gemini-3.1-pro-preview.toml | 4 ++++ providers/vivgrid/models/gpt-5-mini.toml | 5 +++++ providers/vivgrid/models/gpt-5.4-mini.toml | 5 +++++ providers/vivgrid/models/gpt-5.4-nano.toml | 5 +++++ providers/vivgrid/models/gpt-5.4.toml | 5 +++++ providers/vivgrid/models/gpt-5.5.toml | 5 +++++ providers/vultr/provider.toml | 4 ++++ providers/wafer.ai/provider.toml | 5 +++++ providers/wandb/provider.toml | 6 +++++ providers/xiaomi-token-plan-ams/provider.toml | 6 +++++ providers/xiaomi-token-plan-cn/provider.toml | 6 +++++ providers/xiaomi-token-plan-sgp/provider.toml | 6 +++++ providers/xiaomi/provider.toml | 6 +++++ providers/xpersona/provider.toml | 3 +++ providers/zai-coding-plan/provider.toml | 3 +++ providers/zai/models/glm-5.2.toml | 3 +++ providers/zai/provider.toml | 3 +++ providers/zeldoc/models/z-code.toml | 2 ++ .../models/google/gemini-2.5-flash-lite.toml | 2 ++ .../models/google/gemini-2.5-flash.toml | 2 ++ .../zenmux/models/google/gemini-2.5-pro.toml | 2 ++ providers/zenmux/provider.toml | 7 ++++++ providers/zhipuai-coding-plan/provider.toml | 3 +++ providers/zhipuai/models/glm-5.2.toml | 3 +++ providers/zhipuai/provider.toml | 3 +++ 533 files changed, 2155 insertions(+), 6 deletions(-) diff --git a/providers/302ai/provider.toml b/providers/302ai/provider.toml index af2a9ad4f..aa9acd503 100644 --- a/providers/302ai/provider.toml +++ b/providers/302ai/provider.toml @@ -1,5 +1,11 @@ name = "302.AI" env = ["302AI_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# Audited POST https://api.302.ai/v1/chat/completions. The provider's API guide +# documents model/messages only; no reasoning toggle, effort, or numeric budget +# request field is documented. Do not infer passthrough from upstream APIs. +# Sources: +# https://doc.302.ai/ doc = "https://doc.302.ai" api = "https://api.302.ai/v1" diff --git a/providers/abacus/provider.toml b/providers/abacus/provider.toml index dd6f63a45..9ec951182 100644 --- a/providers/abacus/provider.toml +++ b/providers/abacus/provider.toml @@ -1,5 +1,11 @@ name = "Abacus" npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# Audited POST https://routellm.abacus.ai/v1/chat/completions. The provider API +# reference documents no reasoning toggle, effort, or numeric budget request +# field. Do not infer behavior from the routed model developer's API. +# Sources: +# https://abacus.ai/help/api env = ["ABACUS_API_KEY"] doc = "https://abacus.ai/help/api" api = "https://routellm.abacus.ai/v1" diff --git a/providers/abliteration-ai/models/abliterated-model.toml b/providers/abliteration-ai/models/abliterated-model.toml index 8442e6c03..541ef225a 100644 --- a/providers/abliteration-ai/models/abliterated-model.toml +++ b/providers/abliteration-ai/models/abliterated-model.toml @@ -1,4 +1,10 @@ name = "Abliterated Model" +# Reasoning HTTP format (accessed 2026-06-25): +# This model thinks by default. On POST /v1/chat/completions or /v1/messages, +# top-level `thinking: false` skips thinking; omission keeps it enabled. +# Sources: +# https://docs.abliteration.ai/models +# https://docs.abliteration.ai/capabilities/thinking release_date = "2026-01-06" last_updated = "2026-01-06" attachment = true diff --git a/providers/abliteration-ai/provider.toml b/providers/abliteration-ai/provider.toml index f3cbd195c..d6b4982b9 100644 --- a/providers/abliteration-ai/provider.toml +++ b/providers/abliteration-ai/provider.toml @@ -1,5 +1,12 @@ name = "abliteration.ai" env = ["ABLIT_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions and POST /v1/messages: top-level `thinking` is true +# by default; false skips thinking. POST /v1/responses has no thinking toggle. +# No effort or numeric reasoning-budget request field is documented. +# Sources: +# https://docs.abliteration.ai/capabilities/thinking +# https://docs.abliteration.ai/compatibility-matrix api = "https://api.abliteration.ai/v1" doc = "https://docs.abliteration.ai/models" diff --git a/providers/aihubmix/models/claude-opus-4-6.toml b/providers/aihubmix/models/claude-opus-4-6.toml index 06ba09fa1..ab65c0283 100644 --- a/providers/aihubmix/models/claude-opus-4-6.toml +++ b/providers/aihubmix/models/claude-opus-4-6.toml @@ -5,6 +5,7 @@ last_updated = "2026-03-13" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] +# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"|"max"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/claude-opus-4-7.toml b/providers/aihubmix/models/claude-opus-4-7.toml index 99a892d80..39f688aef 100644 --- a/providers/aihubmix/models/claude-opus-4-7.toml +++ b/providers/aihubmix/models/claude-opus-4-7.toml @@ -5,6 +5,7 @@ last_updated = "2026-04-16" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +# Native Messages uses $.thinking.type = "adaptive" and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected. https://docs.aihubmix.com/cn/blogs/Claude-Opus4.7 (accessed 2026-06-25) temperature = false tool_call = true structured_output = true diff --git a/providers/aihubmix/models/claude-sonnet-4-6.toml b/providers/aihubmix/models/claude-sonnet-4-6.toml index 187890742..6b0997a73 100644 --- a/providers/aihubmix/models/claude-sonnet-4-6.toml +++ b/providers/aihubmix/models/claude-sonnet-4-6.toml @@ -5,6 +5,7 @@ last_updated = "2026-03-13" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] +# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. Chat effort "max" maps to native "high". https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/gemini-2.5-flash.toml b/providers/aihubmix/models/gemini-2.5-flash.toml index bda996587..98bf9695a 100644 --- a/providers/aihubmix/models/gemini-2.5-flash.toml +++ b/providers/aihubmix/models/gemini-2.5-flash.toml @@ -5,6 +5,7 @@ last_updated = "2025-06-05" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 0, max = 24_576 }] +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/gemini-2.5-pro.toml b/providers/aihubmix/models/gemini-2.5-pro.toml index f6267c281..204056884 100644 --- a/providers/aihubmix/models/gemini-2.5-pro.toml +++ b/providers/aihubmix/models/gemini-2.5-pro.toml @@ -5,6 +5,7 @@ last_updated = "2025-06-05" attachment = true reasoning = true reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: -1 is dynamic and manual budgets are 128..32768; 0/off is unsupported. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/provider.toml b/providers/aihubmix/provider.toml index 0a925296a..171a1996e 100644 --- a/providers/aihubmix/provider.toml +++ b/providers/aihubmix/provider.toml @@ -1,5 +1,9 @@ name = "AIHubMix" npm = "@aihubmix/ai-sdk-provider" +# Raw Chat: $.reasoning_effort = "none"|"minimal"|"low"|"medium"|"high"|"xhigh"; aliases are $.reasoning.effort and integer $.reasoning.max_tokens. "none" disables models that permit it. https://docs.aihubmix.com/cn/api/unified-inference (accessed 2026-06-25) +# Raw Responses: $.reasoning.effort carries effort; this endpoint has no reasoning-token budget field. https://docs.aihubmix.com/cn/api-reference/openai-compatible/create-a-model-response (accessed 2026-06-25) +# Raw Messages: $.thinking.type = "enabled"|"disabled"|"adaptive"; enabled uses $.thinking.budget_tokens >= 1024, and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max" subject to model support. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-06-25) +# Raw Gemini native: $.generationConfig.thinkingConfig uses integer thinkingBudget (-1 dynamic; 0 off where supported) or string thinkingLevel; model bounds differ below. https://docs.aihubmix.com/cn/api-reference/google-vertex-ai-compatible/generate-content (accessed 2026-06-25) env = ["AIHUBMIX_API_KEY"] doc = "https://docs.aihubmix.com" diff --git a/providers/alibaba-cn/provider.toml b/providers/alibaba-cn/provider.toml index a648a8906..ff9e690f7 100644 --- a/providers/alibaba-cn/provider.toml +++ b/providers/alibaba-cn/provider.toml @@ -1,5 +1,20 @@ name = "Alibaba (China)" env = ["DASHSCOPE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI Chat: POST /compatible-mode/v1/chat/completions; top-level +# `enable_thinking` is true or false and `thinking_budget` is an integer token +# cap (Qwen3 in thinking mode and Kimi only; default is the model maximum). +# Responses: POST /compatible-mode/v1/responses; `reasoning.effort` is none, +# minimal, low, medium (default), or high; no numeric thinking budget is accepted. +# Anthropic: POST /apps/anthropic/v1/messages; `thinking.type` is enabled or +# disabled and `thinking.budget_tokens` is an integer used only when enabled. +# `reasoning_effort` is high or max only for DeepSeek V4 Pro/Flash. +# DashScope text: POST /api/v1/services/aigc/text-generation/generation; put +# `enable_thinking` and `thinking_budget` under `parameters`, not at JSON root. +# Sources: +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://help.aliyun.com/zh/model-studio/compatibility-with-openai-responses-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages doc = "https://www.alibabacloud.com/help/en/model-studio/models" api = "https://dashscope.aliyuncs.com/compatible-mode/v1" \ No newline at end of file diff --git a/providers/alibaba-coding-plan-cn/models/MiniMax-M2.5.toml b/providers/alibaba-coding-plan-cn/models/MiniMax-M2.5.toml index 8a247fe09..e4b16722a 100644 --- a/providers/alibaba-coding-plan-cn/models/MiniMax-M2.5.toml +++ b/providers/alibaba-coding-plan-cn/models/MiniMax-M2.5.toml @@ -1,4 +1,10 @@ name = "MiniMax-M2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model; Alibaba-deployed MiniMax-M2.5 is +# thinking-only. No toggle, effort, or numeric budget request field is documented. +# Sources: +# https://help.aliyun.com/zh/model-studio/coding-plan +# https://help.aliyun.com/zh/model-studio/deep-thinking family = "minimax" release_date = "2026-02-12" last_updated = "2026-02-12" diff --git a/providers/alibaba-coding-plan-cn/models/glm-5.toml b/providers/alibaba-coding-plan-cn/models/glm-5.toml index 21def00b3..29e50e0a8 100644 --- a/providers/alibaba-coding-plan-cn/models/glm-5.toml +++ b/providers/alibaba-coding-plan-cn/models/glm-5.toml @@ -1,4 +1,10 @@ name = "GLM-5" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. Hybrid thinking is on by default and +# top-level `enable_thinking`: true|false toggles it. No effort/budget documented. +# Sources: +# https://help.aliyun.com/zh/model-studio/coding-plan +# https://help.aliyun.com/zh/model-studio/deep-thinking family = "glm" release_date = "2026-02-11" last_updated = "2026-02-11" diff --git a/providers/alibaba-coding-plan-cn/models/kimi-k2.5.toml b/providers/alibaba-coding-plan-cn/models/kimi-k2.5.toml index ac68aecdc..a198e737a 100644 --- a/providers/alibaba-coding-plan-cn/models/kimi-k2.5.toml +++ b/providers/alibaba-coding-plan-cn/models/kimi-k2.5.toml @@ -1,4 +1,11 @@ name = "Kimi K2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. Hybrid thinking uses top-level +# `enable_thinking`: true enables; false (default) disables. No effort or numeric +# budget bound is documented specifically for the plan endpoint. +# Sources: +# https://help.aliyun.com/zh/model-studio/coding-plan +# https://www.alibabacloud.com/help/en/model-studio/kimi-api family = "kimi-k2" release_date = "2026-01-27" last_updated = "2026-01-27" diff --git a/providers/alibaba-coding-plan-cn/models/qwen3-coder-plus.toml b/providers/alibaba-coding-plan-cn/models/qwen3-coder-plus.toml index 7c5b2790b..1a56d6256 100644 --- a/providers/alibaba-coding-plan-cn/models/qwen3-coder-plus.toml +++ b/providers/alibaba-coding-plan-cn/models/qwen3-coder-plus.toml @@ -1,4 +1,9 @@ name = "Qwen3 Coder Plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. The plan docs document no reasoning +# toggle, effort, or numeric budget field for this model. +# Sources: +# https://help.aliyun.com/zh/model-studio/coding-plan family = "qwen" release_date = "2025-07-23" last_updated = "2025-07-23" diff --git a/providers/alibaba-coding-plan-cn/models/qwen3.7-plus.toml b/providers/alibaba-coding-plan-cn/models/qwen3.7-plus.toml index 0fb0ffed8..d03b0720e 100644 --- a/providers/alibaba-coding-plan-cn/models/qwen3.7-plus.toml +++ b/providers/alibaba-coding-plan-cn/models/qwen3.7-plus.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.7-plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. Hybrid thinking is on by default; +# top-level `enable_thinking`: true|false toggles it. The plan docs do not state +# a numeric budget bound or a plan-specific effort control. +# Sources: +# https://help.aliyun.com/zh/model-studio/coding-plan +# https://help.aliyun.com/zh/model-studio/deep-thinking reasoning_options = [{ type = "toggle" }] [cost] diff --git a/providers/alibaba-coding-plan-cn/provider.toml b/providers/alibaba-coding-plan-cn/provider.toml index 2d55b42bb..c244c582a 100644 --- a/providers/alibaba-coding-plan-cn/provider.toml +++ b/providers/alibaba-coding-plan-cn/provider.toml @@ -1,5 +1,15 @@ name = "Alibaba Coding Plan (China)" env = ["ALIBABA_CODING_PLAN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI plan base https://coding.dashscope.aliyuncs.com/v1 uses POST +# /chat/completions with top-level `enable_thinking`: true or false. +# Anthropic plan base https://coding.dashscope.aliyuncs.com/apps/anthropic uses +# POST /v1/messages with `thinking.type`: enabled or disabled and optional +# integer `thinking.budget_tokens`. No plan-specific effort field is documented. +# Sources: +# https://help.aliyun.com/zh/model-studio/coding-plan +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages doc = "https://help.aliyun.com/zh/model-studio/coding-plan" api = "https://coding.dashscope.aliyuncs.com/v1" diff --git a/providers/alibaba-coding-plan/models/MiniMax-M2.5.toml b/providers/alibaba-coding-plan/models/MiniMax-M2.5.toml index 6e4689360..0ab19e457 100644 --- a/providers/alibaba-coding-plan/models/MiniMax-M2.5.toml +++ b/providers/alibaba-coding-plan/models/MiniMax-M2.5.toml @@ -1,4 +1,10 @@ name = "MiniMax-M2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model; Alibaba-deployed MiniMax-M2.5 is +# thinking-only. No toggle, effort, or numeric budget request field is documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/coding-plan +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking family = "minimax" release_date = "2026-02-12" last_updated = "2026-02-12" diff --git a/providers/alibaba-coding-plan/models/glm-5.toml b/providers/alibaba-coding-plan/models/glm-5.toml index 21def00b3..13d1feaa3 100644 --- a/providers/alibaba-coding-plan/models/glm-5.toml +++ b/providers/alibaba-coding-plan/models/glm-5.toml @@ -1,4 +1,10 @@ name = "GLM-5" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. Hybrid thinking is on by default and +# top-level `enable_thinking`: true|false toggles it. No effort/budget documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/coding-plan +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking family = "glm" release_date = "2026-02-11" last_updated = "2026-02-11" diff --git a/providers/alibaba-coding-plan/models/kimi-k2.5.toml b/providers/alibaba-coding-plan/models/kimi-k2.5.toml index 80e643862..d84b08c60 100644 --- a/providers/alibaba-coding-plan/models/kimi-k2.5.toml +++ b/providers/alibaba-coding-plan/models/kimi-k2.5.toml @@ -1,4 +1,11 @@ name = "Kimi K2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. Hybrid thinking uses top-level +# `enable_thinking`: true enables; false (default) disables. No effort or numeric +# budget bound is documented specifically for the plan endpoint. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/coding-plan +# https://www.alibabacloud.com/help/en/model-studio/kimi-api family = "kimi-k2" release_date = "2026-01-27" last_updated = "2026-01-27" diff --git a/providers/alibaba-coding-plan/models/qwen3-coder-plus.toml b/providers/alibaba-coding-plan/models/qwen3-coder-plus.toml index 03677dc81..4a5176a68 100644 --- a/providers/alibaba-coding-plan/models/qwen3-coder-plus.toml +++ b/providers/alibaba-coding-plan/models/qwen3-coder-plus.toml @@ -1,4 +1,9 @@ name = "Qwen3 Coder Plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. The plan docs document no reasoning +# toggle, effort, or numeric budget field for this model. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/coding-plan family = "qwen" release_date = "2025-07-23" last_updated = "2025-07-23" diff --git a/providers/alibaba-coding-plan/models/qwen3.7-plus.toml b/providers/alibaba-coding-plan/models/qwen3.7-plus.toml index 0fb0ffed8..d7ce761e4 100644 --- a/providers/alibaba-coding-plan/models/qwen3.7-plus.toml +++ b/providers/alibaba-coding-plan/models/qwen3.7-plus.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.7-plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Coding Plan exact-string allowlist model. Hybrid thinking is on by default; +# top-level `enable_thinking`: true|false toggles it. The plan docs do not state +# a numeric budget bound or a plan-specific effort control. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/coding-plan +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking reasoning_options = [{ type = "toggle" }] [cost] diff --git a/providers/alibaba-coding-plan/provider.toml b/providers/alibaba-coding-plan/provider.toml index 8c947e861..4af2160f9 100644 --- a/providers/alibaba-coding-plan/provider.toml +++ b/providers/alibaba-coding-plan/provider.toml @@ -1,5 +1,15 @@ name = "Alibaba Coding Plan" env = ["ALIBABA_CODING_PLAN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI plan base https://coding-intl.dashscope.aliyuncs.com/v1 uses POST +# /chat/completions with top-level `enable_thinking`: true or false. +# Anthropic plan base https://coding-intl.dashscope.aliyuncs.com/apps/anthropic +# uses POST /v1/messages with `thinking.type`: enabled or disabled and optional +# integer `thinking.budget_tokens`. No plan-specific effort field is documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/coding-plan +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages doc = "https://www.alibabacloud.com/help/en/model-studio/coding-plan" api = "https://coding-intl.dashscope.aliyuncs.com/v1" diff --git a/providers/alibaba-token-plan-cn/models/MiniMax-M2.5.toml b/providers/alibaba-token-plan-cn/models/MiniMax-M2.5.toml index 32c4a2430..9a14b9501 100644 --- a/providers/alibaba-token-plan-cn/models/MiniMax-M2.5.toml +++ b/providers/alibaba-token-plan-cn/models/MiniMax-M2.5.toml @@ -1,4 +1,9 @@ base_model = "minimax/MiniMax-M2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Alibaba-deployed MiniMax-M2.5 is thinking-only. No toggle, effort, or numeric +# reasoning-budget request field is documented for this model. +# Sources: +# https://help.aliyun.com/zh/model-studio/deep-thinking reasoning_options = [] [interleaved] diff --git a/providers/alibaba-token-plan-cn/models/deepseek-v4-flash.toml b/providers/alibaba-token-plan-cn/models/deepseek-v4-flash.toml index 22af6f9c7..98ddefb93 100644 --- a/providers/alibaba-token-plan-cn/models/deepseek-v4-flash.toml +++ b/providers/alibaba-token-plan-cn/models/deepseek-v4-flash.toml @@ -1,4 +1,11 @@ base_model = "deepseek/deepseek-v4-flash" +# Reasoning HTTP format (accessed 2026-06-25): +# This model toggles with `enable_thinking`: true or false. Chat Completions +# `reasoning_effort` accepts high (default) or max; low/medium map to high and +# xhigh maps to max. Anthropic Messages defaults `reasoning_effort` to max. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deepseek-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages [[reasoning_options]] type = "toggle" diff --git a/providers/alibaba-token-plan-cn/models/deepseek-v4-pro.toml b/providers/alibaba-token-plan-cn/models/deepseek-v4-pro.toml index a5a6fa35c..96343baa1 100644 --- a/providers/alibaba-token-plan-cn/models/deepseek-v4-pro.toml +++ b/providers/alibaba-token-plan-cn/models/deepseek-v4-pro.toml @@ -1,4 +1,11 @@ base_model = "deepseek/deepseek-v4-pro" +# Reasoning HTTP format (accessed 2026-06-25): +# This model toggles with `enable_thinking`: true or false. Chat Completions +# `reasoning_effort` accepts high (default) or max; low/medium map to high and +# xhigh maps to max. Anthropic Messages defaults `reasoning_effort` to max. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deepseek-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages [[reasoning_options]] type = "toggle" diff --git a/providers/alibaba-token-plan-cn/models/kimi-k2.5.toml b/providers/alibaba-token-plan-cn/models/kimi-k2.5.toml index f3d6f6363..62e98617d 100644 --- a/providers/alibaba-token-plan-cn/models/kimi-k2.5.toml +++ b/providers/alibaba-token-plan-cn/models/kimi-k2.5.toml @@ -1,4 +1,10 @@ base_model = "moonshotai/kimi-k2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid model: top-level `enable_thinking` true enables thinking; false is the +# default. `thinking_budget` is an integer token cap; no bounds are documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/kimi-api +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking base_model_omit = ["structured_output"] family = "kimi-k2" attachment = true diff --git a/providers/alibaba-token-plan-cn/models/kimi-k2.6.toml b/providers/alibaba-token-plan-cn/models/kimi-k2.6.toml index c54b81853..c65c9a97f 100644 --- a/providers/alibaba-token-plan-cn/models/kimi-k2.6.toml +++ b/providers/alibaba-token-plan-cn/models/kimi-k2.6.toml @@ -1,4 +1,10 @@ base_model = "moonshotai/kimi-k2.6" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid model: top-level `enable_thinking` true enables thinking; false is the +# default. `thinking_budget` is an integer token cap; no bounds are documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/kimi-api +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking base_model_omit = ["structured_output"] family = "kimi-k2" diff --git a/providers/alibaba-token-plan-cn/models/kimi-k2.7-code.toml b/providers/alibaba-token-plan-cn/models/kimi-k2.7-code.toml index eeef1fe4d..34ea2fec5 100644 --- a/providers/alibaba-token-plan-cn/models/kimi-k2.7-code.toml +++ b/providers/alibaba-token-plan-cn/models/kimi-k2.7-code.toml @@ -1,4 +1,10 @@ base_model = "moonshotai/kimi-k2.7-code" +# Reasoning HTTP format (accessed 2026-06-25): +# Thinking-only model: `enable_thinking` defaults to true and cannot disable +# thinking. `thinking_budget` is an integer token cap; no bounds are documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/kimi-api +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking base_model_omit = ["structured_output"] family = "kimi-k2" reasoning_options = [{ type = "toggle" }, { type = "budget_tokens" }] diff --git a/providers/alibaba-token-plan-cn/models/qwen3.6-flash.toml b/providers/alibaba-token-plan-cn/models/qwen3.6-flash.toml index 56f883884..24f674a32 100644 --- a/providers/alibaba-token-plan-cn/models/qwen3.6-flash.toml +++ b/providers/alibaba-token-plan-cn/models/qwen3.6-flash.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.6-flash" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 81920; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://help.aliyun.com/zh/model-studio/compatibility-with-openai-responses-api [[reasoning_options]] type = "toggle" diff --git a/providers/alibaba-token-plan-cn/models/qwen3.6-plus.toml b/providers/alibaba-token-plan-cn/models/qwen3.6-plus.toml index 54ac24c51..7e6d8d6e7 100644 --- a/providers/alibaba-token-plan-cn/models/qwen3.6-plus.toml +++ b/providers/alibaba-token-plan-cn/models/qwen3.6-plus.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.6-plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 81920; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://help.aliyun.com/zh/model-studio/compatibility-with-openai-responses-api [[reasoning_options]] type = "toggle" diff --git a/providers/alibaba-token-plan-cn/models/qwen3.7-max.toml b/providers/alibaba-token-plan-cn/models/qwen3.7-max.toml index 2368ea771..216ad457d 100644 --- a/providers/alibaba-token-plan-cn/models/qwen3.7-max.toml +++ b/providers/alibaba-token-plan-cn/models/qwen3.7-max.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.7-max" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 262144; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://help.aliyun.com/zh/model-studio/compatibility-with-openai-responses-api [[reasoning_options]] type = "toggle" diff --git a/providers/alibaba-token-plan-cn/models/qwen3.7-plus.toml b/providers/alibaba-token-plan-cn/models/qwen3.7-plus.toml index f0c9690dd..3ac3f3d70 100644 --- a/providers/alibaba-token-plan-cn/models/qwen3.7-plus.toml +++ b/providers/alibaba-token-plan-cn/models/qwen3.7-plus.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.7-plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 262144; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://help.aliyun.com/zh/model-studio/compatibility-with-openai-responses-api [[reasoning_options]] type = "toggle" diff --git a/providers/alibaba-token-plan-cn/provider.toml b/providers/alibaba-token-plan-cn/provider.toml index 5e54e780f..dc8509fb7 100644 --- a/providers/alibaba-token-plan-cn/provider.toml +++ b/providers/alibaba-token-plan-cn/provider.toml @@ -1,5 +1,17 @@ name = "Alibaba Token Plan (China)" env = ["ALIBABA_TOKEN_PLAN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI plan base .../compatible-mode/v1 uses POST /chat/completions with +# top-level `enable_thinking`: true or false and integer `thinking_budget`. +# POST /responses uses `reasoning.effort`: none, minimal, low, medium, or high; +# `thinking_budget` is not accepted there. Anthropic plan base .../apps/anthropic +# uses POST /v1/messages with `thinking.type`: enabled or disabled, integer +# `thinking.budget_tokens`, and DeepSeek V4-only `reasoning_effort`: high or max. +# Sources: +# https://help.aliyun.com/zh/model-studio/token-plan-quickstart +# https://help.aliyun.com/zh/model-studio/deep-thinking +# https://help.aliyun.com/zh/model-studio/compatibility-with-openai-responses-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages doc = "https://www.alibabacloud.com/help/zh/model-studio/token-plan-overview" api = "https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1" diff --git a/providers/alibaba-token-plan/models/MiniMax-M2.5.toml b/providers/alibaba-token-plan/models/MiniMax-M2.5.toml index f313fa080..61e391db8 100644 --- a/providers/alibaba-token-plan/models/MiniMax-M2.5.toml +++ b/providers/alibaba-token-plan/models/MiniMax-M2.5.toml @@ -1,4 +1,9 @@ base_model = "minimax/MiniMax-M2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Alibaba-deployed MiniMax-M2.5 is thinking-only. No toggle, effort, or numeric +# reasoning-budget request field is documented for this model. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking reasoning_options = [] diff --git a/providers/alibaba-token-plan/models/deepseek-v4-flash.toml b/providers/alibaba-token-plan/models/deepseek-v4-flash.toml index 30525f6a1..2c5961fa8 100644 --- a/providers/alibaba-token-plan/models/deepseek-v4-flash.toml +++ b/providers/alibaba-token-plan/models/deepseek-v4-flash.toml @@ -1,4 +1,11 @@ base_model = "deepseek/deepseek-v4-flash" +# Reasoning HTTP format (accessed 2026-06-25): +# This model toggles with `enable_thinking`: true or false. Chat Completions +# `reasoning_effort` accepts high (default) or max; low/medium map to high and +# xhigh maps to max. Anthropic Messages defaults `reasoning_effort` to max. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deepseek-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] diff --git a/providers/alibaba-token-plan/models/deepseek-v4-pro.toml b/providers/alibaba-token-plan/models/deepseek-v4-pro.toml index 7ec1204ff..eef9953fe 100644 --- a/providers/alibaba-token-plan/models/deepseek-v4-pro.toml +++ b/providers/alibaba-token-plan/models/deepseek-v4-pro.toml @@ -1,4 +1,11 @@ base_model = "deepseek/deepseek-v4-pro" +# Reasoning HTTP format (accessed 2026-06-25): +# This model toggles with `enable_thinking`: true or false. Chat Completions +# `reasoning_effort` accepts high (default) or max; low/medium map to high and +# xhigh maps to max. Anthropic Messages defaults `reasoning_effort` to max. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deepseek-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] diff --git a/providers/alibaba-token-plan/models/kimi-k2.5.toml b/providers/alibaba-token-plan/models/kimi-k2.5.toml index 03271f0ea..964c08f56 100644 --- a/providers/alibaba-token-plan/models/kimi-k2.5.toml +++ b/providers/alibaba-token-plan/models/kimi-k2.5.toml @@ -1,4 +1,10 @@ base_model = "moonshotai/kimi-k2.5" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid model: top-level `enable_thinking` true enables thinking; false is the +# default. `thinking_budget` is an integer token cap; no bounds are documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/kimi-api +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking base_model_omit = ["structured_output"] family = "kimi-k2" attachment = true diff --git a/providers/alibaba-token-plan/models/kimi-k2.6.toml b/providers/alibaba-token-plan/models/kimi-k2.6.toml index 00901701d..108df44b7 100644 --- a/providers/alibaba-token-plan/models/kimi-k2.6.toml +++ b/providers/alibaba-token-plan/models/kimi-k2.6.toml @@ -1,4 +1,10 @@ base_model = "moonshotai/kimi-k2.6" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid model: top-level `enable_thinking` true enables thinking; false is the +# default. `thinking_budget` is an integer token cap; no bounds are documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/kimi-api +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking base_model_omit = ["structured_output"] family = "kimi-k2" reasoning_options = [{ type = "toggle" }, { type = "budget_tokens" }] diff --git a/providers/alibaba-token-plan/models/kimi-k2.7-code.toml b/providers/alibaba-token-plan/models/kimi-k2.7-code.toml index eeef1fe4d..34ea2fec5 100644 --- a/providers/alibaba-token-plan/models/kimi-k2.7-code.toml +++ b/providers/alibaba-token-plan/models/kimi-k2.7-code.toml @@ -1,4 +1,10 @@ base_model = "moonshotai/kimi-k2.7-code" +# Reasoning HTTP format (accessed 2026-06-25): +# Thinking-only model: `enable_thinking` defaults to true and cannot disable +# thinking. `thinking_budget` is an integer token cap; no bounds are documented. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/kimi-api +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking base_model_omit = ["structured_output"] family = "kimi-k2" reasoning_options = [{ type = "toggle" }, { type = "budget_tokens" }] diff --git a/providers/alibaba-token-plan/models/qwen3.6-flash.toml b/providers/alibaba-token-plan/models/qwen3.6-flash.toml index ea0391b9d..163b427b7 100644 --- a/providers/alibaba-token-plan/models/qwen3.6-flash.toml +++ b/providers/alibaba-token-plan/models/qwen3.6-flash.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.6-flash" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 81920; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/compatibility-with-openai-responses-api reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 81_920 }] diff --git a/providers/alibaba-token-plan/models/qwen3.6-plus.toml b/providers/alibaba-token-plan/models/qwen3.6-plus.toml index 9a50cea5e..afffb0331 100644 --- a/providers/alibaba-token-plan/models/qwen3.6-plus.toml +++ b/providers/alibaba-token-plan/models/qwen3.6-plus.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.6-plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 81920; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/compatibility-with-openai-responses-api reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 81_920 }] diff --git a/providers/alibaba-token-plan/models/qwen3.7-max.toml b/providers/alibaba-token-plan/models/qwen3.7-max.toml index 30a7d1b2b..48a03f6f8 100644 --- a/providers/alibaba-token-plan/models/qwen3.7-max.toml +++ b/providers/alibaba-token-plan/models/qwen3.7-max.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.7-max" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 262144; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/compatibility-with-openai-responses-api reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] diff --git a/providers/alibaba-token-plan/models/qwen3.7-plus.toml b/providers/alibaba-token-plan/models/qwen3.7-plus.toml index a1e3fe1a5..d81b7888e 100644 --- a/providers/alibaba-token-plan/models/qwen3.7-plus.toml +++ b/providers/alibaba-token-plan/models/qwen3.7-plus.toml @@ -1,4 +1,11 @@ base_model = "alibaba/qwen3.7-plus" +# Reasoning HTTP format (accessed 2026-06-25): +# Hybrid, thinking on by default. Chat/DashScope `enable_thinking`: true|false; +# integer `thinking_budget` maximum 262144; no minimum or disable sentinel is +# documented. Responses instead uses `reasoning.effort` and no numeric budget. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/compatibility-with-openai-responses-api reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] diff --git a/providers/alibaba-token-plan/provider.toml b/providers/alibaba-token-plan/provider.toml index 97b34dae8..97fb48e27 100644 --- a/providers/alibaba-token-plan/provider.toml +++ b/providers/alibaba-token-plan/provider.toml @@ -1,5 +1,17 @@ name = "Alibaba Token Plan" env = ["ALIBABA_TOKEN_PLAN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI plan base .../compatible-mode/v1 uses POST /chat/completions with +# top-level `enable_thinking`: true or false and integer `thinking_budget`. +# POST /responses uses `reasoning.effort`: none, minimal, low, medium, or high; +# `thinking_budget` is not accepted there. Anthropic plan base .../apps/anthropic +# uses POST /v1/messages with `thinking.type`: enabled or disabled, integer +# `thinking.budget_tokens`, and DeepSeek V4-only `reasoning_effort`: high or max. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/token-plan-quickstart +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/compatibility-with-openai-responses-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages doc = "https://www.alibabacloud.com/help/en/model-studio/token-plan-overview" api = "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1" diff --git a/providers/alibaba/provider.toml b/providers/alibaba/provider.toml index 885efe966..498109b6e 100644 --- a/providers/alibaba/provider.toml +++ b/providers/alibaba/provider.toml @@ -1,5 +1,20 @@ name = "Alibaba" env = ["DASHSCOPE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI Chat: POST /compatible-mode/v1/chat/completions; top-level +# `enable_thinking` is true or false and `thinking_budget` is an integer token +# cap (Qwen3 in thinking mode and Kimi only; default is the model maximum). +# Responses: POST /compatible-mode/v1/responses; `reasoning.effort` is none, +# minimal, low, medium (default), or high; no numeric thinking budget is accepted. +# Anthropic: POST /apps/anthropic/v1/messages; `thinking.type` is enabled or +# disabled and `thinking.budget_tokens` is an integer used only when enabled. +# `reasoning_effort` is high or max only for DeepSeek V4 Pro/Flash. +# DashScope text: POST /api/v1/services/aigc/text-generation/generation; put +# `enable_thinking` and `thinking_budget` under `parameters`, not at JSON root. +# Sources: +# https://www.alibabacloud.com/help/en/model-studio/deep-thinking +# https://www.alibabacloud.com/help/en/model-studio/compatibility-with-openai-responses-api +# https://www.alibabacloud.com/help/en/model-studio/anthropic-api-messages doc = "https://www.alibabacloud.com/help/en/model-studio/models" api = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1" \ No newline at end of file diff --git a/providers/ambient/models/moonshotai/kimi-k2.6.toml b/providers/ambient/models/moonshotai/kimi-k2.6.toml index 38a3ddac7..9a348e50b 100644 --- a/providers/ambient/models/moonshotai/kimi-k2.6.toml +++ b/providers/ambient/models/moonshotai/kimi-k2.6.toml @@ -1,3 +1,5 @@ +# Ambient documents no toggle, effort, or budget field for this model; reasoning +# output does not itself establish a control. https://docs.ambient.xyz reasoning_options = [] base_model = "moonshotai/kimi-k2.6" diff --git a/providers/ambient/models/zai-org/GLM-5.1-FP8.toml b/providers/ambient/models/zai-org/GLM-5.1-FP8.toml index 251edab80..27490f91a 100644 --- a/providers/ambient/models/zai-org/GLM-5.1-FP8.toml +++ b/providers/ambient/models/zai-org/GLM-5.1-FP8.toml @@ -1,5 +1,7 @@ base_model = "zhipuai/glm-5.1" name = "GLM 5.1" +# Ambient documents no toggle, effort, or budget field for this model; reasoning +# output does not itself establish a control. https://docs.ambient.xyz reasoning_options = [] [interleaved] diff --git a/providers/ambient/provider.toml b/providers/ambient/provider.toml index 716fbce5d..6d11467ae 100644 --- a/providers/ambient/provider.toml +++ b/providers/ambient/provider.toml @@ -1,5 +1,11 @@ name = "Ambient" env = ["AMBIENT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST https://api.ambient.xyz/v1/chat/completions. Ambient's developer +# surface documents no reasoning toggle, effort, or token-budget request field. +# A reasoning-capable model listing alone does not establish a usable control. +# https://ambient.xyz/developers +# https://docs.ambient.xyz api = "https://api.ambient.xyz/v1" doc = "https://ambient.xyz" diff --git a/providers/anyapi/models/anthropic/claude-haiku-4-5.toml b/providers/anyapi/models/anthropic/claude-haiku-4-5.toml index eedb7fb9c..5a88477f6 100644 --- a/providers/anyapi/models/anthropic/claude-haiku-4-5.toml +++ b/providers/anyapi/models/anthropic/claude-haiku-4-5.toml @@ -1,2 +1,6 @@ base_model = "anthropic/claude-haiku-4-5" +# AnyAPI accepts thinking.type="enabled" with budget_tokens, but documents no +# model-specific 1,024..63,999 bounds or off mapping; these remain upstream facts. +# https://docs.anyapi.ai/guides/parameters +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1_024, max = 63_999 }] diff --git a/providers/anyapi/models/anthropic/claude-opus-4-6.toml b/providers/anyapi/models/anthropic/claude-opus-4-6.toml index 6388ae11a..277f16431 100644 --- a/providers/anyapi/models/anthropic/claude-opus-4-6.toml +++ b/providers/anyapi/models/anthropic/claude-opus-4-6.toml @@ -1,4 +1,8 @@ base_model = "anthropic/claude-opus-4-6" +# AnyAPI accepts thinking.type="enabled" with budget_tokens, but documents no +# model-specific 1,024..127,999 bounds or off mapping; these remain upstream facts. +# https://docs.anyapi.ai/guides/parameters +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1_024, max = 127_999 }] [experimental.modes.fast] diff --git a/providers/anyapi/models/anthropic/claude-opus-4-7.toml b/providers/anyapi/models/anthropic/claude-opus-4-7.toml index cf8b6ea16..3b19885ae 100644 --- a/providers/anyapi/models/anthropic/claude-opus-4-7.toml +++ b/providers/anyapi/models/anthropic/claude-opus-4-7.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-opus-4-7" +# AnyAPI documents generic effort and thinking fields, but no Opus 4.7-specific +# mapping or budget bounds; accepted schema does not prove effective control. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] [experimental.modes.fast] diff --git a/providers/anyapi/models/anthropic/claude-sonnet-4-5.toml b/providers/anyapi/models/anthropic/claude-sonnet-4-5.toml index 1aff685ca..a43b65919 100644 --- a/providers/anyapi/models/anthropic/claude-sonnet-4-5.toml +++ b/providers/anyapi/models/anthropic/claude-sonnet-4-5.toml @@ -1,2 +1,6 @@ base_model = "anthropic/claude-sonnet-4-5" +# AnyAPI accepts thinking.type="enabled" with budget_tokens, but documents no +# model-specific 1,024..63,999 bounds or off mapping; these remain upstream facts. +# https://docs.anyapi.ai/guides/parameters +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1_024, max = 63_999 }] diff --git a/providers/anyapi/models/anthropic/claude-sonnet-4-6.toml b/providers/anyapi/models/anthropic/claude-sonnet-4-6.toml index d893b5bbf..69d77c806 100644 --- a/providers/anyapi/models/anthropic/claude-sonnet-4-6.toml +++ b/providers/anyapi/models/anthropic/claude-sonnet-4-6.toml @@ -1,2 +1,6 @@ base_model = "anthropic/claude-sonnet-4-6" +# AnyAPI accepts thinking.type="enabled" with budget_tokens, but documents no +# model-specific 1,024..63,999 bounds or off mapping; these remain upstream facts. +# https://docs.anyapi.ai/guides/parameters +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1_024, max = 63_999 }] diff --git a/providers/anyapi/models/openai/gpt-5-mini.toml b/providers/anyapi/models/openai/gpt-5-mini.toml index e11353744..bb74733f7 100644 --- a/providers/anyapi/models/openai/gpt-5-mini.toml +++ b/providers/anyapi/models/openai/gpt-5-mini.toml @@ -1,2 +1,5 @@ base_model = "openai/gpt-5-mini" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/models/openai/gpt-5.1.toml b/providers/anyapi/models/openai/gpt-5.1.toml index 61c0e8bea..03df71fe4 100644 --- a/providers/anyapi/models/openai/gpt-5.1.toml +++ b/providers/anyapi/models/openai/gpt-5.1.toml @@ -1,2 +1,5 @@ base_model = "openai/gpt-5.1" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/models/openai/gpt-5.2.toml b/providers/anyapi/models/openai/gpt-5.2.toml index 5bdce2796..e64028af6 100644 --- a/providers/anyapi/models/openai/gpt-5.2.toml +++ b/providers/anyapi/models/openai/gpt-5.2.toml @@ -1,2 +1,5 @@ base_model = "openai/gpt-5.2" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/models/openai/gpt-5.4.toml b/providers/anyapi/models/openai/gpt-5.4.toml index 40db319cb..b5c9108b1 100644 --- a/providers/anyapi/models/openai/gpt-5.4.toml +++ b/providers/anyapi/models/openai/gpt-5.4.toml @@ -1,4 +1,7 @@ base_model = "openai/gpt-5.4" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [experimental.modes.fast] diff --git a/providers/anyapi/models/openai/gpt-5.toml b/providers/anyapi/models/openai/gpt-5.toml index 9b985707d..8e862c0d4 100644 --- a/providers/anyapi/models/openai/gpt-5.toml +++ b/providers/anyapi/models/openai/gpt-5.toml @@ -1,2 +1,5 @@ base_model = "openai/gpt-5" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/models/openai/o3-mini.toml b/providers/anyapi/models/openai/o3-mini.toml index 64bf4a7d1..9da963506 100644 --- a/providers/anyapi/models/openai/o3-mini.toml +++ b/providers/anyapi/models/openai/o3-mini.toml @@ -1,2 +1,5 @@ base_model = "openai/o3-mini" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/models/openai/o3.toml b/providers/anyapi/models/openai/o3.toml index 390b50806..294b5e1f7 100644 --- a/providers/anyapi/models/openai/o3.toml +++ b/providers/anyapi/models/openai/o3.toml @@ -1,2 +1,5 @@ base_model = "openai/o3" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/models/openai/o4-mini.toml b/providers/anyapi/models/openai/o4-mini.toml index 16f1b0a98..27cfe456d 100644 --- a/providers/anyapi/models/openai/o4-mini.toml +++ b/providers/anyapi/models/openai/o4-mini.toml @@ -1,2 +1,5 @@ base_model = "openai/o4-mini" +# AnyAPI accepts low/medium/high generically but documents no model-specific +# effectiveness; unsupported parameters may be silently ignored. +# https://docs.anyapi.ai/guides/parameters reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/anyapi/provider.toml b/providers/anyapi/provider.toml index 558f7c056..656a6a521 100644 --- a/providers/anyapi/provider.toml +++ b/providers/anyapi/provider.toml @@ -1,5 +1,13 @@ name = "AnyAPI" npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST https://api.anyapi.ai/v1/chat/completions accepts top-level +# reasoning_effort = "low" | "medium" | "high" and Anthropic-style +# thinking = { type = "enabled", budget_tokens = }. No budget bounds +# or explicit disabled value are documented. Unsupported model parameters may +# be silently ignored, so schema acceptance is not meaningful model support. +# https://docs.anyapi.ai/guides/parameters +# https://docs.anyapi.ai/openapi.json env = ["ANYAPI_API_KEY"] api = "https://api.anyapi.ai/v1" doc = "https://docs.anyapi.ai" diff --git a/providers/atomic-chat/provider.toml b/providers/atomic-chat/provider.toml index 5dca660b0..9ba59563f 100644 --- a/providers/atomic-chat/provider.toml +++ b/providers/atomic-chat/provider.toml @@ -1,5 +1,13 @@ name = "Atomic Chat" env = ["ATOMIC_CHAT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST http://127.0.0.1:1337/v1/chat/completions. Current source accepts +# chat_template_kwargs.enable_thinking = true | false for template-supported +# models. The former process-level --reasoning-budget accepted -1 (unlimited) +# or 0 (disabled), but was removed; no effort or token-budget HTTP field is +# documented. Request-schema acceptance still depends on the loaded backend. +# https://github.com/AtomicBot-ai/Atomic-Chat/commit/92703bceb2c65e3218f81a15b4e9058f171d4ff2 +# https://github.com/AtomicBot-ai/Atomic-Chat/commit/742e731e966b1b59cedad36301e32fc7613e4bce api = "http://127.0.0.1:1337/v1" doc = "https://atomic.chat" diff --git a/providers/auriko/models/claude-opus-4-6.toml b/providers/auriko/models/claude-opus-4-6.toml index fca6b1fd9..16f9db992 100644 --- a/providers/auriko/models/claude-opus-4-6.toml +++ b/providers/auriko/models/claude-opus-4-6.toml @@ -1,5 +1,8 @@ base_model = "anthropic/claude-opus-4-6" +# Claude 4.6 uses adaptive thinking; Auriko maps effort and "off" to its native +# control. No caller-selected token budget is exposed. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }] [cost] diff --git a/providers/auriko/models/claude-opus-4-7.toml b/providers/auriko/models/claude-opus-4-7.toml index a3b31532c..90bc5baaa 100644 --- a/providers/auriko/models/claude-opus-4-7.toml +++ b/providers/auriko/models/claude-opus-4-7.toml @@ -1,5 +1,8 @@ base_model = "anthropic/claude-opus-4-7" +# Auriko's schema accepts these levels and "off", but the provider support table +# does not yet document an Opus 4.7-specific translation. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [cost] diff --git a/providers/auriko/models/claude-sonnet-4-6.toml b/providers/auriko/models/claude-sonnet-4-6.toml index ce334e952..14075c5eb 100644 --- a/providers/auriko/models/claude-sonnet-4-6.toml +++ b/providers/auriko/models/claude-sonnet-4-6.toml @@ -1,5 +1,8 @@ base_model = "anthropic/claude-sonnet-4-6" +# Claude 4.6 uses adaptive thinking; Auriko maps effort and "off" to its native +# control. No caller-selected token budget is exposed. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }] [cost] diff --git a/providers/auriko/models/deepseek-v4-flash.toml b/providers/auriko/models/deepseek-v4-flash.toml index 339163753..1fe58c298 100644 --- a/providers/auriko/models/deepseek-v4-flash.toml +++ b/providers/auriko/models/deepseek-v4-flash.toml @@ -1,5 +1,8 @@ base_model = "deepseek/deepseek-v4-flash" +# Also aliased as deepseek-chat and deepseek-reasoner; aliases are not modes. +# Effort controls a derived budget and "off" requests non-thinking output. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [interleaved] diff --git a/providers/auriko/models/deepseek-v4-pro.toml b/providers/auriko/models/deepseek-v4-pro.toml index bdfc22a74..e51acaff2 100644 --- a/providers/auriko/models/deepseek-v4-pro.toml +++ b/providers/auriko/models/deepseek-v4-pro.toml @@ -1,5 +1,7 @@ base_model = "deepseek/deepseek-v4-pro" +# Auriko derives a provider thinking budget from effort; "off" disables it. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [interleaved] diff --git a/providers/auriko/models/gemini-2.5-flash.toml b/providers/auriko/models/gemini-2.5-flash.toml index ffcf5badb..305f46f65 100644 --- a/providers/auriko/models/gemini-2.5-flash.toml +++ b/providers/auriko/models/gemini-2.5-flash.toml @@ -1,5 +1,8 @@ base_model = "google/gemini-2.5-flash" +# Auriko derives a Gemini thinking budget from effort; "off" disables thinking. +# Provider extension keys google/google_ai/googleai/gemini are aliases. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [cost] diff --git a/providers/auriko/models/gemini-2.5-pro.toml b/providers/auriko/models/gemini-2.5-pro.toml index d2e862877..1ce6068e0 100644 --- a/providers/auriko/models/gemini-2.5-pro.toml +++ b/providers/auriko/models/gemini-2.5-pro.toml @@ -1,5 +1,8 @@ base_model = "google/gemini-2.5-pro" +# Auriko derives a Gemini thinking budget from effort. Although schema-level +# "off" exists, this model metadata does not claim a toggle. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [cost] diff --git a/providers/auriko/models/gemini-3.1-pro-preview.toml b/providers/auriko/models/gemini-3.1-pro-preview.toml index cb2df505c..20ebe41df 100644 --- a/providers/auriko/models/gemini-3.1-pro-preview.toml +++ b/providers/auriko/models/gemini-3.1-pro-preview.toml @@ -1,5 +1,8 @@ base_model = "google/gemini-3.1-pro-preview" +# Gemini 3.x maps only low/medium/high to thinking levels; xhigh/max normalize +# to high. Provider extension keys google/google_ai/googleai/gemini are aliases. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/auriko/models/glm-5.1.toml b/providers/auriko/models/glm-5.1.toml index 092d07a63..9965876a8 100644 --- a/providers/auriko/models/glm-5.1.toml +++ b/providers/auriko/models/glm-5.1.toml @@ -1,5 +1,8 @@ base_model = "zhipuai/glm-5.1" +# Auriko's schema accepts reasoning_effort, but its provider support table gives +# no GLM 5.1 mapping; no meaningful toggle, effort, or budget is documented. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [] [interleaved] diff --git a/providers/auriko/models/grok-4.3.toml b/providers/auriko/models/grok-4.3.toml index 6db39adc6..3a1302b7d 100644 --- a/providers/auriko/models/grok-4.3.toml +++ b/providers/auriko/models/grok-4.3.toml @@ -1,5 +1,8 @@ base_model = "xai/grok-4.3" +# Grok 4.3 meaningfully supports native low/medium/high; xhigh/max normalize to +# high, and "off" is Auriko's normalized disable value. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/auriko/models/kimi-k2.5.toml b/providers/auriko/models/kimi-k2.5.toml index c49970fcf..a5ef19922 100644 --- a/providers/auriko/models/kimi-k2.5.toml +++ b/providers/auriko/models/kimi-k2.5.toml @@ -1,5 +1,7 @@ base_model = "moonshotai/kimi-k2.5" +# Auriko maps effort to Moonshot's native control; "off" disables thinking. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [interleaved] diff --git a/providers/auriko/models/kimi-k2.6.toml b/providers/auriko/models/kimi-k2.6.toml index dc6ea1eb2..5cb511f9d 100644 --- a/providers/auriko/models/kimi-k2.6.toml +++ b/providers/auriko/models/kimi-k2.6.toml @@ -1,5 +1,7 @@ base_model = "moonshotai/kimi-k2.6" +# Auriko maps effort to Moonshot's native control; "off" disables thinking. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] [interleaved] diff --git a/providers/auriko/models/minimax-m2-7-highspeed.toml b/providers/auriko/models/minimax-m2-7-highspeed.toml index eb0913123..1a5e6f03b 100644 --- a/providers/auriko/models/minimax-m2-7-highspeed.toml +++ b/providers/auriko/models/minimax-m2-7-highspeed.toml @@ -1,5 +1,8 @@ base_model = "minimax/MiniMax-M2.7-highspeed" +# Auriko documents M2-series reasoning as built in and drops reasoning_effort; +# schema acceptance is therefore not a meaningful control for this model. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [] [cost] diff --git a/providers/auriko/models/minimax-m2-7.toml b/providers/auriko/models/minimax-m2-7.toml index 0ed99cd4e..5c7a73f26 100644 --- a/providers/auriko/models/minimax-m2-7.toml +++ b/providers/auriko/models/minimax-m2-7.toml @@ -1,5 +1,8 @@ base_model = "minimax/MiniMax-M2.7" +# Auriko documents M2-series reasoning as built in and drops reasoning_effort; +# schema acceptance is therefore not a meaningful control for this model. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [] [cost] diff --git a/providers/auriko/models/qwen-3.6-plus.toml b/providers/auriko/models/qwen-3.6-plus.toml index e94330eb7..b2e5b3409 100644 --- a/providers/auriko/models/qwen-3.6-plus.toml +++ b/providers/auriko/models/qwen-3.6-plus.toml @@ -1,5 +1,8 @@ base_model = "alibaba/qwen3.6-plus" +# Auriko's schema accepts reasoning_effort, but its provider support table gives +# no Qwen 3.6 mapping; no meaningful toggle, effort, or budget is documented. +# https://docs.auriko.ai/guides/extensions-and-thinking reasoning_options = [] [cost] diff --git a/providers/auriko/provider.toml b/providers/auriko/provider.toml index 2bc98ae9a..5d71ab5a5 100644 --- a/providers/auriko/provider.toml +++ b/providers/auriko/provider.toml @@ -1,5 +1,14 @@ name = "Auriko" env = ["AURIKO_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST https://api.auriko.ai/v1/chat/completions accepts top-level +# reasoning_effort = "low" | "medium" | "high" | "xhigh" | "max" | "off"; +# "off" disables thinking. Auriko translates values per model/provider and may +# clamp them with a warning. No caller-specified reasoning-token budget exists; +# translated budgets are internal, and budget fields in extensions are replaced. +# Provider extension aliases google, google_ai, googleai, and gemini are equal. +# https://docs.auriko.ai/guides/extensions-and-thinking +# https://docs.auriko.ai/api-reference/chat-completions api = "https://api.auriko.ai/v1" doc = "https://docs.auriko.ai" diff --git a/providers/azure-cognitive-services/models/claude-haiku-4-5.toml b/providers/azure-cognitive-services/models/claude-haiku-4-5.toml index fd48b389d..980596ac7 100644 --- a/providers/azure-cognitive-services/models/claude-haiku-4-5.toml +++ b/providers/azure-cognitive-services/models/claude-haiku-4-5.toml @@ -4,6 +4,9 @@ release_date = "2025-11-18" last_updated = "2025-11-18" attachment = true reasoning = true +# POST /anthropic/v1/messages uses thinking.type="enabled" and integer +# thinking.budget_tokens >=1024; budget must be below max_tokens. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "budget_tokens", min = 1_024 }] temperature = true knowledge = "2025-02-31" diff --git a/providers/azure-cognitive-services/models/claude-opus-4-1.toml b/providers/azure-cognitive-services/models/claude-opus-4-1.toml index ef28f1423..782a9bb71 100644 --- a/providers/azure-cognitive-services/models/claude-opus-4-1.toml +++ b/providers/azure-cognitive-services/models/claude-opus-4-1.toml @@ -4,6 +4,9 @@ release_date = "2025-11-18" last_updated = "2025-11-18" attachment = true reasoning = true +# POST /anthropic/v1/messages uses thinking.type="enabled" and integer +# thinking.budget_tokens >=1024; budget must be below max_tokens. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "budget_tokens", min = 1_024 }] temperature = true knowledge = "2025-03-31" diff --git a/providers/azure-cognitive-services/models/claude-opus-4-5.toml b/providers/azure-cognitive-services/models/claude-opus-4-5.toml index 7782c8ebc..daac288be 100644 --- a/providers/azure-cognitive-services/models/claude-opus-4-5.toml +++ b/providers/azure-cognitive-services/models/claude-opus-4-5.toml @@ -4,6 +4,9 @@ release_date = "2025-11-24" last_updated = "2025-08-01" attachment = true reasoning = true +# POST /anthropic/v1/messages accepts thinking.budget_tokens >=1024 and effort +# low/medium/high for this model; a raw budget must remain below max_tokens. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1_024 }] temperature = true tool_call = true diff --git a/providers/azure-cognitive-services/models/claude-opus-4-6.toml b/providers/azure-cognitive-services/models/claude-opus-4-6.toml index efbf83942..e7e8e9f6f 100644 --- a/providers/azure-cognitive-services/models/claude-opus-4-6.toml +++ b/providers/azure-cognitive-services/models/claude-opus-4-6.toml @@ -4,6 +4,9 @@ release_date = "2026-02-05" last_updated = "2026-02-05" attachment = true reasoning = true +# POST /anthropic/v1/messages supports adaptive thinking with output_config.effort +# low/medium/high/max; manual thinking.budget_tokens is integer >=1024. +# https://docs.anthropic.com/en/docs/build-with-claude/adaptive-thinking reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] temperature = true tool_call = true diff --git a/providers/azure-cognitive-services/models/claude-opus-4-8.toml b/providers/azure-cognitive-services/models/claude-opus-4-8.toml index f204efc52..27df61c4d 100644 --- a/providers/azure-cognitive-services/models/claude-opus-4-8.toml +++ b/providers/azure-cognitive-services/models/claude-opus-4-8.toml @@ -1,6 +1,9 @@ base_model = "anthropic/claude-opus-4-8" temperature = true knowledge = "2025-12-31" +# POST /anthropic/v1/messages uses adaptive thinking with output_config.effort +# low/medium/high/max; Azure catalog availability alone is not a control. +# https://docs.anthropic.com/en/docs/build-with-claude/adaptive-thinking reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "max"] }] [cost] diff --git a/providers/azure-cognitive-services/models/claude-sonnet-4-5.toml b/providers/azure-cognitive-services/models/claude-sonnet-4-5.toml index 0cc5cdf92..3b52d643e 100644 --- a/providers/azure-cognitive-services/models/claude-sonnet-4-5.toml +++ b/providers/azure-cognitive-services/models/claude-sonnet-4-5.toml @@ -4,6 +4,9 @@ release_date = "2025-11-18" last_updated = "2025-11-18" attachment = true reasoning = true +# POST /anthropic/v1/messages uses thinking.type="enabled" and integer +# thinking.budget_tokens >=1024; budget must be below max_tokens. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking reasoning_options = [{ type = "budget_tokens", min = 1_024 }] temperature = true knowledge = "2025-07-31" diff --git a/providers/azure-cognitive-services/models/gpt-5.4-mini.toml b/providers/azure-cognitive-services/models/gpt-5.4-mini.toml index ef57e7413..1c7f3277c 100644 --- a/providers/azure-cognitive-services/models/gpt-5.4-mini.toml +++ b/providers/azure-cognitive-services/models/gpt-5.4-mini.toml @@ -1,4 +1,7 @@ base_model = "openai/gpt-5.4-mini" +# Azure OpenAI Chat uses top-level reasoning_effort; this model accepts +# none/low/medium/high/xhigh. No separate token-budget field is documented. +# https://learn.microsoft.com/en-us/azure/foundry/openai/how-to/reasoning reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] name = "GPT-5.4 Mini" diff --git a/providers/azure-cognitive-services/models/gpt-5.4-nano.toml b/providers/azure-cognitive-services/models/gpt-5.4-nano.toml index 90c6d2271..864b7e0dc 100644 --- a/providers/azure-cognitive-services/models/gpt-5.4-nano.toml +++ b/providers/azure-cognitive-services/models/gpt-5.4-nano.toml @@ -1,4 +1,7 @@ base_model = "openai/gpt-5.4-nano" +# Azure OpenAI Chat uses top-level reasoning_effort; this model accepts +# none/low/medium/high/xhigh. No separate token-budget field is documented. +# https://learn.microsoft.com/en-us/azure/foundry/openai/how-to/reasoning reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] name = "GPT-5.4 Nano" diff --git a/providers/azure-cognitive-services/models/gpt-5.4-pro.toml b/providers/azure-cognitive-services/models/gpt-5.4-pro.toml index 113307480..92c995b76 100644 --- a/providers/azure-cognitive-services/models/gpt-5.4-pro.toml +++ b/providers/azure-cognitive-services/models/gpt-5.4-pro.toml @@ -1,4 +1,7 @@ base_model = "openai/gpt-5.4-pro" +# Azure documents this Responses-only model's reasoning.effort subset as +# medium/high/xhigh; no toggle or reasoning-token budget field is exposed. +# https://learn.microsoft.com/en-us/azure/foundry/openai/how-to/reasoning reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] [cost] diff --git a/providers/azure-cognitive-services/models/gpt-5.4.toml b/providers/azure-cognitive-services/models/gpt-5.4.toml index 230eac09d..760d8e165 100644 --- a/providers/azure-cognitive-services/models/gpt-5.4.toml +++ b/providers/azure-cognitive-services/models/gpt-5.4.toml @@ -1,4 +1,7 @@ base_model = "openai/gpt-5.4" +# Azure OpenAI Chat uses top-level reasoning_effort; this model accepts +# none/low/medium/high/xhigh. No separate token-budget field is documented. +# https://learn.microsoft.com/en-us/azure/foundry/openai/how-to/reasoning reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] [cost] diff --git a/providers/azure-cognitive-services/models/kimi-k2.5.toml b/providers/azure-cognitive-services/models/kimi-k2.5.toml index 02d01e449..1c87db59d 100644 --- a/providers/azure-cognitive-services/models/kimi-k2.5.toml +++ b/providers/azure-cognitive-services/models/kimi-k2.5.toml @@ -1,4 +1,7 @@ base_model = "moonshotai/kimi-k2.5" +# Azure documents POST /models/chat/completions and reasoning output, but no +# toggle, effort, or budget request field for Kimi K2.5. Catalog modes are not +# request controls. https://learn.microsoft.com/en-us/azure/ai-foundry/foundry-models/how-to/use-chat-reasoning reasoning_options = [{ type = "toggle" }] temperature = true interleaved = true diff --git a/providers/azure-cognitive-services/models/kimi-k2.6.toml b/providers/azure-cognitive-services/models/kimi-k2.6.toml index 91b91acd0..df3b012f0 100644 --- a/providers/azure-cognitive-services/models/kimi-k2.6.toml +++ b/providers/azure-cognitive-services/models/kimi-k2.6.toml @@ -1,5 +1,8 @@ base_model = "moonshotai/kimi-k2.6" attachment = false +# Azure documents POST /models/chat/completions and reasoning output, but no +# toggle, effort, or budget request field for Kimi K2.6. Catalog modes are not +# request controls. https://learn.microsoft.com/en-us/azure/ai-foundry/foundry-models/how-to/use-chat-reasoning reasoning_options = [{ type = "toggle" }] interleaved = true diff --git a/providers/azure-cognitive-services/provider.toml b/providers/azure-cognitive-services/provider.toml index 0d62a8578..cbdc60f3c 100644 --- a/providers/azure-cognitive-services/provider.toml +++ b/providers/azure-cognitive-services/provider.toml @@ -1,4 +1,12 @@ name = "Azure Cognitive Services" env = ["AZURE_COGNITIVE_SERVICES_RESOURCE_NAME", "AZURE_COGNITIVE_SERVICES_API_KEY"] npm = "@ai-sdk/azure" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# Azure OpenAI Chat: POST https://.openai.azure.com/openai/v1/chat/completions +# uses top-level reasoning_effort; model-specific values are documented below. +# Anthropic Messages: POST https://.services.ai.azure.com/anthropic/v1/messages +# uses thinking.type and thinking.budget_tokens. Azure model "modes" alone are +# not request controls. Azure documents no common toggle or budget across APIs. +# https://learn.microsoft.com/en-us/azure/foundry/openai/how-to/reasoning +# https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure doc = "https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models" \ No newline at end of file diff --git a/providers/azure/models/deepseek-v4-flash.toml b/providers/azure/models/deepseek-v4-flash.toml index 65b64b031..2e6430d70 100644 --- a/providers/azure/models/deepseek-v4-flash.toml +++ b/providers/azure/models/deepseek-v4-flash.toml @@ -1,3 +1,6 @@ +# Direct endpoint: POST /models/chat/completions?api-version=2024-05-01-preview. +# Its published request schema documents no toggle, effort, or reasoning budget. +# https://learn.microsoft.com/en-us/rest/api/microsoftfoundry/model-inference/get-chat-completions/get-chat-completions (accessed 2026-06-25) base_model = "deepseek/deepseek-v4-flash" name = "DeepSeek-V4-Flash" tool_call = false diff --git a/providers/azure/models/deepseek-v4-pro.toml b/providers/azure/models/deepseek-v4-pro.toml index cf558fbcf..dd0cb3bb8 100644 --- a/providers/azure/models/deepseek-v4-pro.toml +++ b/providers/azure/models/deepseek-v4-pro.toml @@ -1,3 +1,6 @@ +# Direct endpoint: POST /models/chat/completions?api-version=2024-05-01-preview. +# Its published request schema documents no toggle, effort, or reasoning budget. +# https://learn.microsoft.com/en-us/rest/api/microsoftfoundry/model-inference/get-chat-completions/get-chat-completions (accessed 2026-06-25) base_model = "deepseek/deepseek-v4-pro" name = "DeepSeek-V4-Pro" tool_call = false diff --git a/providers/azure/models/kimi-k2.5.toml b/providers/azure/models/kimi-k2.5.toml index a509e99c1..5146f436f 100644 --- a/providers/azure/models/kimi-k2.5.toml +++ b/providers/azure/models/kimi-k2.5.toml @@ -1,3 +1,6 @@ +# Direct endpoint: POST /models/chat/completions?api-version=2024-05-01-preview. +# Its published request schema documents no toggle, effort, or reasoning budget. +# https://learn.microsoft.com/en-us/rest/api/microsoftfoundry/model-inference/get-chat-completions/get-chat-completions (accessed 2026-06-25) name = "Kimi K2.5" family = "kimi-k2" release_date = "2026-02-06" diff --git a/providers/azure/models/kimi-k2.6.toml b/providers/azure/models/kimi-k2.6.toml index c935f7854..9320476bc 100644 --- a/providers/azure/models/kimi-k2.6.toml +++ b/providers/azure/models/kimi-k2.6.toml @@ -1,3 +1,6 @@ +# Direct endpoint: POST /models/chat/completions?api-version=2024-05-01-preview. +# Its published request schema documents no toggle, effort, or reasoning budget. +# https://learn.microsoft.com/en-us/rest/api/microsoftfoundry/model-inference/get-chat-completions/get-chat-completions (accessed 2026-06-25) name = "Kimi K2.6" family = "kimi-k2" release_date = "2026-04-22" diff --git a/providers/bailing/models/Ring-1T.toml b/providers/bailing/models/Ring-1T.toml index aff36e583..6cdb30a3d 100644 --- a/providers/bailing/models/Ring-1T.toml +++ b/providers/bailing/models/Ring-1T.toml @@ -4,6 +4,8 @@ release_date = "2025-10" last_updated = "2025-10" attachment = false reasoning = true +# The Bailing request surface documents no toggle, effort, or budget field for +# Ring-1T. https://alipaytbox.yuque.com/sxs0ba/ling/intro reasoning_options = [] temperature = true knowledge = "2024-06" diff --git a/providers/bailing/provider.toml b/providers/bailing/provider.toml index 67ce97af3..cd2b32360 100644 --- a/providers/bailing/provider.toml +++ b/providers/bailing/provider.toml @@ -1,5 +1,10 @@ name = "Bailing" env = ["BAILING_API_TOKEN"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (source accessed 2026-06-25): +# POST https://api.tbox.cn/api/llm/v1/chat/completions. The provider request +# surface documents no reasoning toggle, effort, or token-budget request field; +# Ring-1T being a reasoning model does not by itself establish such a control. +# https://alipaytbox.yuque.com/sxs0ba/ling/intro doc = "https://alipaytbox.yuque.com/sxs0ba/ling/intro" api = "https://api.tbox.cn/api/llm/v1/chat/completions" \ No newline at end of file diff --git a/providers/baseten/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/baseten/models/MiniMaxAI/MiniMax-M2.5.toml index 017e543eb..ec172aa88 100644 --- a/providers/baseten/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/baseten/models/MiniMaxAI/MiniMax-M2.5.toml @@ -5,6 +5,9 @@ release_date = "2026-02-12" last_updated = "2026-02-12" attachment = false reasoning = true +# This deprecated model is absent from Baseten's current reasoning support table; +# no toggle, effort, or budget request field is documented for it. +# https://docs.baseten.co/inference/model-apis/reasoning reasoning_options = [] temperature = true tool_call = true diff --git a/providers/baseten/models/deepseek-ai/DeepSeek-V3.1.toml b/providers/baseten/models/deepseek-ai/DeepSeek-V3.1.toml index 5df7a4c86..ce2c3b517 100644 --- a/providers/baseten/models/deepseek-ai/DeepSeek-V3.1.toml +++ b/providers/baseten/models/deepseek-ai/DeepSeek-V3.1.toml @@ -5,6 +5,9 @@ release_date = "2025-08-25" last_updated = "2025-08-25" attachment = false reasoning = true +# This deprecated model is absent from Baseten's current reasoning support table; +# no toggle, effort, or budget request field is documented for it. +# https://docs.baseten.co/inference/model-apis/reasoning reasoning_options = [] temperature = true tool_call = true diff --git a/providers/baseten/models/deepseek-ai/DeepSeek-V4-Pro.toml b/providers/baseten/models/deepseek-ai/DeepSeek-V4-Pro.toml index cb412cbd2..4f830f2c5 100644 --- a/providers/baseten/models/deepseek-ai/DeepSeek-V4-Pro.toml +++ b/providers/baseten/models/deepseek-ai/DeepSeek-V4-Pro.toml @@ -4,6 +4,9 @@ name = "Deepseek V4 Pro" [interleaved] field = "reasoning_content" +# Baseten documents top-level reasoning_effort = low | medium | high | xhigh +# for DeepSeek V4 Pro; no toggle or reasoning-token budget field is documented. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "effort" values = ["low", "medium", "high", "xhigh"] diff --git a/providers/baseten/models/moonshotai/Kimi-K2.5.toml b/providers/baseten/models/moonshotai/Kimi-K2.5.toml index 41ca7f3c7..04ed4c383 100644 --- a/providers/baseten/models/moonshotai/Kimi-K2.5.toml +++ b/providers/baseten/models/moonshotai/Kimi-K2.5.toml @@ -10,6 +10,9 @@ structured_output = true knowledge = "2025-12" open_weights = true +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/moonshotai/Kimi-K2.6.toml b/providers/baseten/models/moonshotai/Kimi-K2.6.toml index 48f8aa10e..a27ad5a6b 100644 --- a/providers/baseten/models/moonshotai/Kimi-K2.6.toml +++ b/providers/baseten/models/moonshotai/Kimi-K2.6.toml @@ -10,6 +10,9 @@ structured_output = true knowledge = "2025-01" open_weights = true +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/moonshotai/Kimi-K2.7-Code.toml b/providers/baseten/models/moonshotai/Kimi-K2.7-Code.toml index f68038a8c..b2117d153 100644 --- a/providers/baseten/models/moonshotai/Kimi-K2.7-Code.toml +++ b/providers/baseten/models/moonshotai/Kimi-K2.7-Code.toml @@ -1,4 +1,7 @@ base_model = "moonshotai/kimi-k2.7-code" +# Baseten documents opt-in via chat_template_args.enable_thinking=true, but no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning temperature = true [cost] diff --git a/providers/baseten/models/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B.toml b/providers/baseten/models/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B.toml index 91d59311a..e02ffc884 100644 --- a/providers/baseten/models/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B.toml +++ b/providers/baseten/models/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B.toml @@ -2,6 +2,9 @@ base_model = "nvidia/nemotron-3-ultra-550b-a55b" name = "Nemotron Ultra" structured_output = true +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/nvidia/Nemotron-120B-A12B.toml b/providers/baseten/models/nvidia/Nemotron-120B-A12B.toml index 873e0c338..d5c973e4a 100644 --- a/providers/baseten/models/nvidia/Nemotron-120B-A12B.toml +++ b/providers/baseten/models/nvidia/Nemotron-120B-A12B.toml @@ -3,6 +3,9 @@ name = "Nemotron Super" structured_output = true knowledge = "2026-02" +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/openai/gpt-oss-120b.toml b/providers/baseten/models/openai/gpt-oss-120b.toml index 492dfe3f9..655b85082 100644 --- a/providers/baseten/models/openai/gpt-oss-120b.toml +++ b/providers/baseten/models/openai/gpt-oss-120b.toml @@ -10,6 +10,9 @@ structured_output = true knowledge = "2025-08" open_weights = true +# Baseten documents top-level reasoning_effort = low | medium | high for GPT OSS +# 120B; reasoning is otherwise enabled by default. No toggle or budget exists. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/baseten/models/zai-org/GLM-4.7.toml b/providers/baseten/models/zai-org/GLM-4.7.toml index 988f7550e..f31f0014b 100644 --- a/providers/baseten/models/zai-org/GLM-4.7.toml +++ b/providers/baseten/models/zai-org/GLM-4.7.toml @@ -10,6 +10,9 @@ structured_output = true knowledge = "2025-04" open_weights = true +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/zai-org/GLM-5.1.toml b/providers/baseten/models/zai-org/GLM-5.1.toml index 8d5b02979..265e8f07f 100644 --- a/providers/baseten/models/zai-org/GLM-5.1.toml +++ b/providers/baseten/models/zai-org/GLM-5.1.toml @@ -1,6 +1,9 @@ base_model = "zhipuai/glm-5.1" name = "GLM 5.1" +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/zai-org/GLM-5.2.toml b/providers/baseten/models/zai-org/GLM-5.2.toml index 79794c795..1a092c7d8 100644 --- a/providers/baseten/models/zai-org/GLM-5.2.toml +++ b/providers/baseten/models/zai-org/GLM-5.2.toml @@ -1,6 +1,9 @@ base_model = "zhipuai/glm-5.2" name = "GLM 5.2" +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/models/zai-org/GLM-5.toml b/providers/baseten/models/zai-org/GLM-5.toml index 6f3edf574..46868f4b8 100644 --- a/providers/baseten/models/zai-org/GLM-5.toml +++ b/providers/baseten/models/zai-org/GLM-5.toml @@ -10,6 +10,9 @@ structured_output = true knowledge = "2026-01" open_weights = true +# Opt in with chat_template_args.enable_thinking=true. Baseten documents no +# explicit false behavior, effort values, or reasoning-token budget. +# https://docs.baseten.co/inference/model-apis/reasoning [[reasoning_options]] type = "toggle" diff --git a/providers/baseten/provider.toml b/providers/baseten/provider.toml index d27803c8c..0d187c0ca 100644 --- a/providers/baseten/provider.toml +++ b/providers/baseten/provider.toml @@ -1,5 +1,14 @@ name = "Baseten" npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST https://inference.baseten.co/v1/chat/completions uses top-level +# chat_template_args.enable_thinking = true for documented opt-in models and +# top-level reasoning_effort for DeepSeek V4 Pro and GPT OSS 120B. No explicit +# false toggle behavior or reasoning-token budget field/bounds are documented. +# The strict OpenAPI omits reasoning_effort but allows chat_template_args; +# model support is established by the model-specific reasoning guide, not schema. +# https://docs.baseten.co/inference/model-apis/reasoning +# https://docs.baseten.co/reference/inference-api/chat-completions doc = "https://docs.baseten.co/inference/model-apis/overview" api = "https://inference.baseten.co/v1" env = ["BASETEN_API_KEY"] diff --git a/providers/berget/models/google/gemma-4-31B-it.toml b/providers/berget/models/google/gemma-4-31B-it.toml index b8eb01e58..7684ce1c8 100644 --- a/providers/berget/models/google/gemma-4-31B-it.toml +++ b/providers/berget/models/google/gemma-4-31B-it.toml @@ -4,6 +4,8 @@ release_date = "2026-04-02" last_updated = "2026-04-02" attachment = true reasoning = true +# Berget documents no meaningful toggle, effort subset, or budget for this model. +# https://api.berget.ai/openapi.json reasoning_options = [] temperature = true knowledge = "2025-12" diff --git a/providers/berget/models/meta-llama/Llama-3.3-70B-Instruct.toml b/providers/berget/models/meta-llama/Llama-3.3-70B-Instruct.toml index 15fa89eba..9ed2e8a28 100644 --- a/providers/berget/models/meta-llama/Llama-3.3-70B-Instruct.toml +++ b/providers/berget/models/meta-llama/Llama-3.3-70B-Instruct.toml @@ -4,6 +4,8 @@ release_date = "2025-04-27" last_updated = "2025-04-27" attachment = false reasoning = true +# Berget documents no meaningful toggle, effort subset, or budget for this model. +# https://api.berget.ai/openapi.json reasoning_options = [] temperature = true knowledge = "2023-12" diff --git a/providers/berget/models/mistralai/Mistral-Medium-3.5-128B.toml b/providers/berget/models/mistralai/Mistral-Medium-3.5-128B.toml index 0b23de9c0..e508adcbb 100644 --- a/providers/berget/models/mistralai/Mistral-Medium-3.5-128B.toml +++ b/providers/berget/models/mistralai/Mistral-Medium-3.5-128B.toml @@ -4,6 +4,9 @@ release_date = "2026-04-29" last_updated = "2026-04-29" attachment = true reasoning = true +# Berget's schema accepts none/low/medium/high, but documents no Mistral-specific +# mapping; schema acceptance alone does not establish meaningful support. +# https://api.berget.ai/openapi.json reasoning_options = [{ type = "effort", values = ["none", "high"] }] temperature = true knowledge = "2026-04" diff --git a/providers/berget/models/mistralai/Mistral-Small-3.2-24B-Instruct-2506.toml b/providers/berget/models/mistralai/Mistral-Small-3.2-24B-Instruct-2506.toml index 18ce89621..f7ef31b2a 100644 --- a/providers/berget/models/mistralai/Mistral-Small-3.2-24B-Instruct-2506.toml +++ b/providers/berget/models/mistralai/Mistral-Small-3.2-24B-Instruct-2506.toml @@ -4,6 +4,8 @@ release_date = "2025-10-01" last_updated = "2025-10-01" attachment = false reasoning = true +# Berget documents no meaningful toggle, effort subset, or budget for this model. +# https://api.berget.ai/openapi.json reasoning_options = [] temperature = true knowledge = "2025-09" diff --git a/providers/berget/models/moonshotai/Kimi-K2.6.toml b/providers/berget/models/moonshotai/Kimi-K2.6.toml index 039a4aee4..6d5e74554 100644 --- a/providers/berget/models/moonshotai/Kimi-K2.6.toml +++ b/providers/berget/models/moonshotai/Kimi-K2.6.toml @@ -1,4 +1,7 @@ base_model = "moonshotai/kimi-k2.6" +# Kimi K2.6 can disable reasoning with reasoning_effort="none" or +# thinking.type="disabled"; enabled/adaptive are also schema-accepted. +# https://api.berget.ai/openapi.json reasoning_options = [{ type = "toggle" }] release_date = "2026-05-07" last_updated = "2026-05-07" diff --git a/providers/berget/models/openai/gpt-oss-120b.toml b/providers/berget/models/openai/gpt-oss-120b.toml index e62081bff..c08519224 100644 --- a/providers/berget/models/openai/gpt-oss-120b.toml +++ b/providers/berget/models/openai/gpt-oss-120b.toml @@ -4,6 +4,9 @@ release_date = "2025-08-05" last_updated = "2025-08-05" attachment = false reasoning = true +# The request schema accepts reasoning_effort low/medium/high, but publishes no +# GPT OSS-specific behavioral mapping or reasoning-token budget. +# https://api.berget.ai/openapi.json reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true knowledge = "2025-08" diff --git a/providers/berget/models/zai-org/GLM-4.7.toml b/providers/berget/models/zai-org/GLM-4.7.toml index 6dce2eba5..367c990d5 100644 --- a/providers/berget/models/zai-org/GLM-4.7.toml +++ b/providers/berget/models/zai-org/GLM-4.7.toml @@ -4,6 +4,8 @@ release_date = "2026-01-19" last_updated = "2026-01-19" attachment = false reasoning = true +# Berget documents no meaningful toggle, effort subset, or budget for this model. +# https://api.berget.ai/openapi.json reasoning_options = [] temperature = true knowledge = "2025-12" diff --git a/providers/berget/models/zai-org/GLM-5.2.toml b/providers/berget/models/zai-org/GLM-5.2.toml index b75d1cbb9..628b8848e 100644 --- a/providers/berget/models/zai-org/GLM-5.2.toml +++ b/providers/berget/models/zai-org/GLM-5.2.toml @@ -1,4 +1,7 @@ base_model = "zhipuai/glm-5.2" +# Berget's schema accepts none/low/medium/high, but documents no GLM-specific +# mapping; schema acceptance alone does not establish meaningful support. +# https://api.berget.ai/openapi.json reasoning_options = [{ type = "effort", values = ["none", "high"] }] [cost] diff --git a/providers/berget/provider.toml b/providers/berget/provider.toml index b93c3265d..c472cf16e 100644 --- a/providers/berget/provider.toml +++ b/providers/berget/provider.toml @@ -1,5 +1,12 @@ name = "Berget.AI" env = ["BERGET_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (source accessed 2026-06-25): +# POST https://api.berget.ai/v1/chat/completions accepts top-level +# reasoning_effort = "none" | "low" | "medium" | "high". For Kimi K2.6 it +# also accepts thinking.type = "disabled" | "enabled" | "adaptive" and maps +# effort "none" to disabled. No reasoning-token budget field is documented. +# Schema acceptance is broader than the meaningful per-model subsets below. +# https://api.berget.ai/openapi.json api = "https://api.berget.ai/v1" doc = "https://api.berget.ai" \ No newline at end of file diff --git a/providers/cerebras/models/gpt-oss-120b.toml b/providers/cerebras/models/gpt-oss-120b.toml index 0712c5640..dbe21d7ab 100644 --- a/providers/cerebras/models/gpt-oss-120b.toml +++ b/providers/cerebras/models/gpt-oss-120b.toml @@ -1,4 +1,9 @@ name = "GPT OSS 120B" +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# reasoning_effort = "low"|"medium"|"high"; default is "medium". +# Sources: +# https://inference-docs.cerebras.ai/capabilities/reasoning#gpt-oss-reasoning-effort +# https://inference-docs.cerebras.ai/models/openai-oss family = "gpt-oss" release_date = "2025-08-05" last_updated = "2026-06-10" diff --git a/providers/cerebras/models/zai-glm-4.7.toml b/providers/cerebras/models/zai-glm-4.7.toml index 624827c9c..6bc1b9450 100644 --- a/providers/cerebras/models/zai-glm-4.7.toml +++ b/providers/cerebras/models/zai-glm-4.7.toml @@ -1,4 +1,11 @@ name = "Z.AI GLM-4.7" +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# Reasoning defaults on; reasoning_effort = "none" disables it. The alternative +# disable_reasoning = true|false is deprecated and will be removed after +# 2026-07-21. No positive effort levels or token budget are documented. +# Sources: +# https://inference-docs.cerebras.ai/capabilities/reasoning#glm-reasoning-effort-and-disable-reasoning +# https://inference-docs.cerebras.ai/models/zai-glm-47 release_date = "2026-01-07" last_updated = "2026-06-10" attachment = false diff --git a/providers/cerebras/provider.toml b/providers/cerebras/provider.toml index b2258c4b0..cfb8082b7 100644 --- a/providers/cerebras/provider.toml +++ b/providers/cerebras/provider.toml @@ -1,4 +1,12 @@ name = "Cerebras" env = ["CEREBRAS_API_KEY"] npm = "@ai-sdk/cerebras" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI Chat: POST https://api.cerebras.ai/v1/chat/completions. +# Effort/toggle: top-level reasoning_effort, with accepted values depending on +# the model. Token budget is not documented in the complete request schema. +# Sources: +# https://inference-docs.cerebras.ai/api-reference/chat-completions +# https://inference-docs.cerebras.ai/capabilities/reasoning +# https://inference-docs.cerebras.ai/resources/openai doc = "https://inference-docs.cerebras.ai/models/overview" diff --git a/providers/chutes/provider.toml b/providers/chutes/provider.toml index d2e519c69..94ede8654 100644 --- a/providers/chutes/provider.toml +++ b/providers/chutes/provider.toml @@ -1,5 +1,12 @@ name = "Chutes" env = ["CHUTES_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST https://llm.chutes.ai/v1/chat/completions. GET /v1/models marks models +# with supported_features including "reasoning", but its advertised sampling +# parameters contain no toggle, effort, or budget field. The reasoning guide's +# request surface likewise documents none; reasoning capability is not control. +# https://llm.chutes.ai/v1/models +# https://chutes.ai/docs/guides/reasoning-models api = "https://llm.chutes.ai/v1" doc = "https://llm.chutes.ai/v1/models" \ No newline at end of file diff --git a/providers/clarifai/models/openai/chat-completion/models/gpt-oss-120b-high-throughput.toml b/providers/clarifai/models/openai/chat-completion/models/gpt-oss-120b-high-throughput.toml index c6aff5202..f7f0e84f2 100644 --- a/providers/clarifai/models/openai/chat-completion/models/gpt-oss-120b-high-throughput.toml +++ b/providers/clarifai/models/openai/chat-completion/models/gpt-oss-120b-high-throughput.toml @@ -4,6 +4,9 @@ release_date = "2025-08-05" last_updated = "2026-02-25" attachment = false reasoning = true +# Clarifai's POST /v2/ext/openai/v1/chat/completions request surface documents +# no reasoning_effort field; upstream model support does not prove forwarding. +# https://docs.clarifai.com/compute/inference/open-ai reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/clarifai/models/openai/chat-completion/models/gpt-oss-20b.toml b/providers/clarifai/models/openai/chat-completion/models/gpt-oss-20b.toml index 16a083f89..38a707c3e 100644 --- a/providers/clarifai/models/openai/chat-completion/models/gpt-oss-20b.toml +++ b/providers/clarifai/models/openai/chat-completion/models/gpt-oss-20b.toml @@ -4,6 +4,9 @@ release_date = "2025-08-05" last_updated = "2025-12-12" attachment = false reasoning = true +# Clarifai's POST /v2/ext/openai/v1/chat/completions request surface documents +# no reasoning_effort field; upstream model support does not prove forwarding. +# https://docs.clarifai.com/compute/inference/open-ai reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/clarifai/provider.toml b/providers/clarifai/provider.toml index af69e9691..fa525b2dc 100644 --- a/providers/clarifai/provider.toml +++ b/providers/clarifai/provider.toml @@ -1,5 +1,11 @@ name = "Clarifai" env = ["CLARIFAI_PAT"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (source accessed 2026-06-25): +# POST https://api.clarifai.com/v2/ext/openai/v1/chat/completions. Clarifai's +# OpenAI-compatible request examples and documented parameter surface expose no +# reasoning toggle, effort, or token-budget field. A hosted model's own ability +# to reason therefore does not establish a Clarifai request control. +# https://docs.clarifai.com/compute/inference/open-ai api = "https://api.clarifai.com/v2/ext/openai/v1" doc = "https://docs.clarifai.com/compute/inference/" diff --git a/providers/claudinio/models/claudinio.toml b/providers/claudinio/models/claudinio.toml index c69fa7152..427e5b3f7 100644 --- a/providers/claudinio/models/claudinio.toml +++ b/providers/claudinio/models/claudinio.toml @@ -1,4 +1,7 @@ name = "Claudinio" +# The configured OpenAI effort shape is `reasoning_effort = low|medium|high`, +# but the published request list documents no reasoning control or budget. +# https://claudin.io/docs/api-reference/ (accessed 2026-06-25) release_date = "2026-05-12" last_updated = "2026-06-02" attachment = true diff --git a/providers/claudinio/provider.toml b/providers/claudinio/provider.toml index 0680aedff..47b80dbba 100644 --- a/providers/claudinio/provider.toml +++ b/providers/claudinio/provider.toml @@ -1,5 +1,9 @@ name = "Claudinio" env = ["CLAUDINIO_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP routes are POST `/v1/chat/completions`, `/v1/responses`, and +# `/v1/messages`. The documented request fields do not include a reasoning +# toggle, effort, or token-budget field. +# https://claudin.io/docs/api-reference/ (accessed 2026-06-25) doc = "https://claudin.io" api = "https://api.claudin.io/v1" diff --git a/providers/cloudferro-sherlock/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/cloudferro-sherlock/models/MiniMaxAI/MiniMax-M2.5.toml index 7dedfe5d5..c5ac2f3c2 100644 --- a/providers/cloudferro-sherlock/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/cloudferro-sherlock/models/MiniMaxAI/MiniMax-M2.5.toml @@ -1,4 +1,7 @@ name = "MiniMax-M2.5" +# Sherlock's published Chat request surface documents no toggle, effort, or +# reasoning-token budget for this reasoning-only model. +# https://docs.sherlock.cloudferro.com/ (accessed 2026-06-25) family = "minimax" release_date = "2026-03-05" last_updated = "2026-03-05" diff --git a/providers/cloudferro-sherlock/models/openai/gpt-oss-120b.toml b/providers/cloudferro-sherlock/models/openai/gpt-oss-120b.toml index c2f7a7bc9..3047e230a 100644 --- a/providers/cloudferro-sherlock/models/openai/gpt-oss-120b.toml +++ b/providers/cloudferro-sherlock/models/openai/gpt-oss-120b.toml @@ -1,4 +1,7 @@ name = "OpenAI GPT OSS 120B" +# The configured OpenAI effort shape is `reasoning_effort = low|medium|high`, +# but Sherlock documents no reasoning control or token budget. +# https://docs.sherlock.cloudferro.com/ (accessed 2026-06-25) family = "gpt-oss" release_date = "2025-08-28" last_updated = "2025-08-28" diff --git a/providers/cloudferro-sherlock/provider.toml b/providers/cloudferro-sherlock/provider.toml index b89eeb368..b59fa8730 100644 --- a/providers/cloudferro-sherlock/provider.toml +++ b/providers/cloudferro-sherlock/provider.toml @@ -1,5 +1,9 @@ name = "CloudFerro Sherlock" env = ["CLOUDFERRO_SHERLOCK_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP is OpenAI Chat at POST `/openai/v1/chat/completions`. Sherlock's +# public request documentation does not document a shared reasoning toggle, +# effort, or token-budget field. +# https://docs.sherlock.cloudferro.com/ (accessed 2026-06-25) api = "https://api-sherlock.cloudferro.com/openai/v1/" doc = "https://docs.sherlock.cloudferro.com/" diff --git a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.5.toml b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.5.toml index b9b7d0d2e..72a8895d7 100644 --- a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.5.toml +++ b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.5.toml @@ -1,3 +1,6 @@ +# Raw Workers AI input uses `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.enable_thinking = true|false`. +# https://developers.cloudflare.com/workers-ai/models/kimi-k2.5/sync-input.json (accessed 2026-06-25) name = "Kimi K2.5" family = "kimi-k2" release_date = "2026-01-27" diff --git a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.6.toml b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.6.toml index c7d0ad3c4..88b5c1c3e 100644 --- a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.6.toml +++ b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/moonshotai/kimi-k2.6.toml @@ -1,3 +1,6 @@ +# Raw Workers AI input uses `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.thinking = true|false`. +# https://developers.cloudflare.com/workers-ai/models/kimi-k2.6/sync-input.json (accessed 2026-06-25) name = "Kimi K2.6" family = "kimi-k2" release_date = "2026-04-20" diff --git a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/nvidia/nemotron-3-120b-a12b.toml b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/nvidia/nemotron-3-120b-a12b.toml index 1d17bb8b3..ff3096436 100644 --- a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/nvidia/nemotron-3-120b-a12b.toml +++ b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/nvidia/nemotron-3-120b-a12b.toml @@ -1,3 +1,6 @@ +# Raw Workers AI input uses `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.enable_thinking = true|false`. +# https://developers.cloudflare.com/workers-ai/models/nemotron-3-120b-a12b/sync-input.json (accessed 2026-06-25) name = "Nemotron 3 Super 120B" base_model = "nvidia/nemotron-3-super-120b-a12b" family = "nemotron" diff --git a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/zai-org/glm-4.7-flash.toml b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/zai-org/glm-4.7-flash.toml index 5ecba78e3..a667664ee 100644 --- a/providers/cloudflare-ai-gateway/models/workers-ai/@cf/zai-org/glm-4.7-flash.toml +++ b/providers/cloudflare-ai-gateway/models/workers-ai/@cf/zai-org/glm-4.7-flash.toml @@ -1,3 +1,6 @@ +# Raw Workers AI input uses `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.enable_thinking = true|false`. +# https://developers.cloudflare.com/workers-ai/models/glm-4.7-flash/sync-input.json (accessed 2026-06-25) name = "GLM-4.7-Flash" family = "glm-flash" release_date = "2026-01-19" diff --git a/providers/cloudflare-ai-gateway/provider.toml b/providers/cloudflare-ai-gateway/provider.toml index 2fec3c9ec..215163491 100644 --- a/providers/cloudflare-ai-gateway/provider.toml +++ b/providers/cloudflare-ai-gateway/provider.toml @@ -1,4 +1,15 @@ name = "Cloudflare AI Gateway" env = ["CLOUDFLARE_API_TOKEN", "CLOUDFLARE_ACCOUNT_ID", "CLOUDFLARE_GATEWAY_ID"] npm = "ai-gateway-provider" +# Provider-native routes preserve upstream bodies: OpenAI Chat/Responses use +# `reasoning_effort`/`reasoning.effort`; Anthropic Messages uses +# `thinking.type`, `thinking.budget_tokens`, and `output_config.effort`. +# https://developers.cloudflare.com/ai-gateway/usage/providers/openai/ (accessed 2026-06-25) +# https://developers.cloudflare.com/ai-gateway/usage/providers/anthropic/ (accessed 2026-06-25) +# Unified Workers AI and third-party routes are POST `/ai/run`, +# `/ai/v1/chat/completions`, `/ai/v1/responses`, and `/ai/v1/messages`; +# `/ai/run` puts each model's native input schema under `input`. +# https://developers.cloudflare.com/ai-gateway/usage/rest-api/ (accessed 2026-06-25) +# Workers AI model input fields are defined by each model's raw schema. +# https://developers.cloudflare.com/workers-ai/models/ (accessed 2026-06-25) doc = "https://developers.cloudflare.com/ai-gateway/" diff --git a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b.toml b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b.toml index 3bcc13d1a..2bcd41037 100644 --- a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b.toml @@ -1,4 +1,7 @@ base_model = "deepseek/deepseek-r1" +# Native `/ai/run` schema documents no reasoning toggle, effort, or token +# budget for this reasoning-only model. +# https://developers.cloudflare.com/workers-ai/models/deepseek-r1-distill-qwen-32b/sync-input.json (accessed 2026-06-25) name = "Deepseek R1 Distill Qwen 32B" reasoning_options = [] tool_call = false diff --git a/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml b/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml index a685122a3..64f70cc24 100644 --- a/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml +++ b/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml @@ -1,4 +1,7 @@ base_model = "google/gemma-4-26b-a4b-it" +# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.thinking = true|false`; no budget is documented. +# https://developers.cloudflare.com/workers-ai/models/gemma-4-26b-a4b-it/sync-input.json (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] interleaved = true diff --git a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml index 671fc7c94..4b3643a7f 100644 --- a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml +++ b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml @@ -1,4 +1,7 @@ base_model = "moonshotai/kimi-k2.6" +# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.thinking = true|false`; no budget is documented. +# https://developers.cloudflare.com/workers-ai/models/kimi-k2.6/sync-input.json (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] [interleaved] diff --git a/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml b/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml index ce23e20eb..68a867316 100644 --- a/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml @@ -1,4 +1,7 @@ base_model = "nvidia/nemotron-3-super-120b-a12b" +# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. +# https://developers.cloudflare.com/workers-ai/models/nemotron-3-120b-a12b/sync-input.json (accessed 2026-06-25) name = "Nemotron 3 Super 120B" structured_output = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-120b.toml b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-120b.toml index b0ae1dcb4..51374acfc 100644 --- a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-120b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-120b.toml @@ -1,4 +1,7 @@ name = "GPT OSS 120B" +# Native `/ai/run` schema has no reasoning control; OpenAI Chat/Responses use +# `reasoning_effort = low|medium|high`. No token budget is documented. +# https://developers.cloudflare.com/workers-ai/models/gpt-oss-120b/sync-input.json (accessed 2026-06-25) family = "gpt-oss" release_date = "2025-08-05" last_updated = "2025-08-05" diff --git a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml index db92cc15b..9b45508b9 100644 --- a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml @@ -1,4 +1,7 @@ name = "GPT OSS 20B" +# Native `/ai/run` schema has no reasoning control; OpenAI Chat/Responses use +# `reasoning_effort = low|medium|high`. No token budget is documented. +# https://developers.cloudflare.com/workers-ai/models/gpt-oss-20b/sync-input.json (accessed 2026-06-25) family = "gpt-oss" release_date = "2025-08-05" last_updated = "2025-08-05" diff --git a/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3-30b-a3b-fp8.toml b/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3-30b-a3b-fp8.toml index 44fae4c42..cb3a10b59 100644 --- a/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3-30b-a3b-fp8.toml +++ b/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3-30b-a3b-fp8.toml @@ -1,4 +1,7 @@ name = "Qwen3 30B A3b fp8" +# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. +# https://developers.cloudflare.com/workers-ai/models/qwen3-30b-a3b-fp8/sync-input.json (accessed 2026-06-25) family = "qwen" release_date = "2025-04-30" last_updated = "2025-04-30" diff --git a/providers/cloudflare-workers-ai/models/@cf/qwen/qwq-32b.toml b/providers/cloudflare-workers-ai/models/@cf/qwen/qwq-32b.toml index b3f9d098d..eb5970c85 100644 --- a/providers/cloudflare-workers-ai/models/@cf/qwen/qwq-32b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/qwen/qwq-32b.toml @@ -1,4 +1,7 @@ name = "Qwq 32B" +# Native `/ai/run` schema documents no reasoning toggle, effort, or token +# budget for this reasoning-only model. +# https://developers.cloudflare.com/workers-ai/models/qwq-32b/sync-input.json (accessed 2026-06-25) family = "qwen" release_date = "2025-03-05" last_updated = "2025-03-05" diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml index ae57f8db0..8ec1d37d4 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml @@ -1,4 +1,7 @@ name = "GLM-4.7-Flash" +# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and +# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. +# https://developers.cloudflare.com/workers-ai/models/glm-4.7-flash/sync-input.json (accessed 2026-06-25) family = "glm-flash" release_date = "2026-01-19" last_updated = "2026-01-19" diff --git a/providers/cloudflare-workers-ai/provider.toml b/providers/cloudflare-workers-ai/provider.toml index 935f6a03c..71031fec4 100644 --- a/providers/cloudflare-workers-ai/provider.toml +++ b/providers/cloudflare-workers-ai/provider.toml @@ -1,5 +1,13 @@ name = "Cloudflare Workers AI" env = ["CLOUDFLARE_ACCOUNT_ID", "CLOUDFLARE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw native HTTP is POST `/client/v4/accounts/{account_id}/ai/run/{model}`; +# each model's `sync-input.json` is the authoritative request schema. +# OpenAI-compatible POST `/ai/v1/chat/completions` and `/ai/v1/responses` use +# `reasoning_effort` or `reasoning.effort`; native template toggles are +# model-specific `chat_template_kwargs.thinking|enable_thinking` booleans. +# https://developers.cloudflare.com/workers-ai/get-started/rest-api/ (accessed 2026-06-25) +# https://developers.cloudflare.com/workers-ai/configuration/open-ai-compatibility/ (accessed 2026-06-25) +# https://developers.cloudflare.com/workers-ai/models/ (accessed 2026-06-25) doc = "https://developers.cloudflare.com/workers-ai/models/" api = "https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1" diff --git a/providers/cohere/models/command-a-plus-05-2026.toml b/providers/cohere/models/command-a-plus-05-2026.toml index 9256e11f3..ca814ca18 100644 --- a/providers/cohere/models/command-a-plus-05-2026.toml +++ b/providers/cohere/models/command-a-plus-05-2026.toml @@ -1,4 +1,12 @@ base_model = "cohere/command-a-plus-05-2026" +# Model-specific reasoning HTTP restrictions (accessed 2026-06-25): +# Native thinking.token_budget must be positive; leave at least 1K output tokens. +# OpenAI compatibility instead accepts reasoning_effort = "none"|"high" and +# does not document token-budget control. +# Sources: +# https://docs.cohere.com/docs/command-a-plus +# https://docs.cohere.com/docs/reasoning +# https://docs.cohere.com/docs/compatibility-api [[reasoning_options]] type = "toggle" diff --git a/providers/cohere/models/command-a-reasoning-08-2025.toml b/providers/cohere/models/command-a-reasoning-08-2025.toml index 7482ba812..92c6118f1 100644 --- a/providers/cohere/models/command-a-reasoning-08-2025.toml +++ b/providers/cohere/models/command-a-reasoning-08-2025.toml @@ -1,4 +1,12 @@ name = "Command A Reasoning" +# Model-specific reasoning HTTP restrictions (accessed 2026-06-25): +# Native thinking.token_budget must be positive; leave at least 1K output tokens. +# OpenAI compatibility instead accepts reasoning_effort = "none"|"high" and +# does not document token-budget control. +# Sources: +# https://docs.cohere.com/docs/command-a-reasoning +# https://docs.cohere.com/docs/reasoning +# https://docs.cohere.com/docs/compatibility-api family = "command-a" release_date = "2025-08-21" last_updated = "2025-08-21" diff --git a/providers/cohere/models/north-mini-code-1-0.toml b/providers/cohere/models/north-mini-code-1-0.toml index f02561b7e..e6c97eaf2 100644 --- a/providers/cohere/models/north-mini-code-1-0.toml +++ b/providers/cohere/models/north-mini-code-1-0.toml @@ -1,4 +1,9 @@ base_model = "cohere/north-mini-code-1-0" +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# OpenAI compatibility reasoning_effort = "none"|"high"; "low" and "medium" +# are explicitly unsupported. Token-budget control is not documented. +# Sources: +# https://docs.cohere.com/docs/compatibility-api [[reasoning_options]] type = "effort" diff --git a/providers/cohere/provider.toml b/providers/cohere/provider.toml index dd0399677..8981dcfac 100644 --- a/providers/cohere/provider.toml +++ b/providers/cohere/provider.toml @@ -1,4 +1,15 @@ name = "Cohere" env = ["COHERE_API_KEY"] npm = "@ai-sdk/cohere" +# Reasoning HTTP format (accessed 2026-06-25): +# Native Chat: POST https://api.cohere.com/v2/chat. Toggle: +# thinking.type = "enabled"|"disabled". Budget: thinking.token_budget = a +# positive integer. Effort is not documented on this request surface. +# OpenAI Chat: POST https://api.cohere.ai/compatibility/v1/chat/completions. +# reasoning_effort = "none"|"high" maps to disabled|enabled thinking; +# token budget is not documented on this request surface. +# Sources: +# https://docs.cohere.com/reference/chat +# https://docs.cohere.com/docs/reasoning +# https://docs.cohere.com/docs/compatibility-api doc = "https://docs.cohere.com/docs/models" diff --git a/providers/cortecs/models/claude-4-5-sonnet.toml b/providers/cortecs/models/claude-4-5-sonnet.toml index 91e2a2010..4a6dee819 100644 --- a/providers/cortecs/models/claude-4-5-sonnet.toml +++ b/providers/cortecs/models/claude-4-5-sonnet.toml @@ -1,4 +1,7 @@ name = "Claude 4.5 Sonnet" +# Cortecs maps `reasoning_effort = low|medium|high` and +# `thinking.budget_tokens >= 1024`; unsupported fields may be silently ignored. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "claude-sonnet" release_date = "2025-09-29" last_updated = "2025-09-29" diff --git a/providers/cortecs/models/claude-4-6-sonnet.toml b/providers/cortecs/models/claude-4-6-sonnet.toml index 6eee03104..980bb2556 100644 --- a/providers/cortecs/models/claude-4-6-sonnet.toml +++ b/providers/cortecs/models/claude-4-6-sonnet.toml @@ -1,4 +1,7 @@ name = "Claude Sonnet 4.6" +# Cortecs maps `reasoning_effort = low|medium|high` and +# `thinking.budget_tokens >= 1024`; unsupported fields may be silently ignored. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "claude-sonnet" release_date = "2026-02-17" last_updated = "2026-03-13" diff --git a/providers/cortecs/models/claude-haiku-4-5.toml b/providers/cortecs/models/claude-haiku-4-5.toml index 27a7d543d..cf3931f14 100644 --- a/providers/cortecs/models/claude-haiku-4-5.toml +++ b/providers/cortecs/models/claude-haiku-4-5.toml @@ -1,4 +1,7 @@ name = "Claude Haiku 4.5" +# Cortecs maps `reasoning_effort = low|medium|high` and +# `thinking.budget_tokens >= 1024`; unsupported fields may be silently ignored. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "claude-haiku" release_date = "2025-10-15" last_updated = "2025-10-15" diff --git a/providers/cortecs/models/claude-opus4-5.toml b/providers/cortecs/models/claude-opus4-5.toml index f858097f1..032816377 100644 --- a/providers/cortecs/models/claude-opus4-5.toml +++ b/providers/cortecs/models/claude-opus4-5.toml @@ -1,4 +1,7 @@ name = "Claude Opus 4.5" +# Cortecs maps `reasoning_effort = low|medium|high` and +# `thinking.budget_tokens >= 1024`; unsupported fields may be silently ignored. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "claude-opus" release_date = "2025-11-24" last_updated = "2025-11-24" diff --git a/providers/cortecs/models/claude-opus4-6.toml b/providers/cortecs/models/claude-opus4-6.toml index d7ad730fb..0ff756a36 100644 --- a/providers/cortecs/models/claude-opus4-6.toml +++ b/providers/cortecs/models/claude-opus4-6.toml @@ -1,4 +1,7 @@ name = "Claude Opus 4.6" +# Cortecs maps `reasoning_effort = low|medium|high` and +# `thinking.budget_tokens >= 1024`; unsupported fields may be silently ignored. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "claude-opus" release_date = "2026-02-05" last_updated = "2026-03-13" diff --git a/providers/cortecs/models/claude-opus4-7.toml b/providers/cortecs/models/claude-opus4-7.toml index 00b2bf2e4..bbe5a40a4 100644 --- a/providers/cortecs/models/claude-opus4-7.toml +++ b/providers/cortecs/models/claude-opus4-7.toml @@ -1,4 +1,7 @@ name = "Claude Opus 4.7" +# Cortecs maps `reasoning_effort = low|medium|high` to +# `output_config.effort`; no explicit budget is exposed for this model. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "claude-opus" release_date = "2026-04-16" last_updated = "2026-04-16" diff --git a/providers/cortecs/models/claude-opus4-8.toml b/providers/cortecs/models/claude-opus4-8.toml index 871ad11bf..2a78c94e5 100644 --- a/providers/cortecs/models/claude-opus4-8.toml +++ b/providers/cortecs/models/claude-opus4-8.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-opus-4-8" +# Cortecs maps `reasoning_effort = low|medium|high` to +# `output_config.effort`; no explicit budget is exposed for this model. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/cortecs/models/deepseek-v4-flash.toml b/providers/cortecs/models/deepseek-v4-flash.toml index 5701c8b69..073727138 100644 --- a/providers/cortecs/models/deepseek-v4-flash.toml +++ b/providers/cortecs/models/deepseek-v4-flash.toml @@ -1,4 +1,6 @@ base_model = "deepseek/deepseek-v4-flash" +# Cortecs Chat maps `reasoning_effort = low|medium|high`; no budget is exposed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) base_model_omit = ["structured_output"] reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/cortecs/models/deepseek-v4-pro.toml b/providers/cortecs/models/deepseek-v4-pro.toml index 143db4167..74f1cba5d 100644 --- a/providers/cortecs/models/deepseek-v4-pro.toml +++ b/providers/cortecs/models/deepseek-v4-pro.toml @@ -1,4 +1,6 @@ base_model = "deepseek/deepseek-v4-pro" +# Cortecs Chat maps `reasoning_effort = low|medium|high`; no budget is exposed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) base_model_omit = ["structured_output"] reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/cortecs/models/glm-5.2.toml b/providers/cortecs/models/glm-5.2.toml index 42ad7cfd1..305e7fbb9 100644 --- a/providers/cortecs/models/glm-5.2.toml +++ b/providers/cortecs/models/glm-5.2.toml @@ -1,4 +1,6 @@ base_model = "zhipuai/glm-5.2" +# Cortecs Chat maps `reasoning_effort = high|max`; other efforts are not listed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["high", "max"]}] [interleaved] diff --git a/providers/cortecs/models/gpt-5.4.toml b/providers/cortecs/models/gpt-5.4.toml index f0ed27f63..9dbc108e7 100644 --- a/providers/cortecs/models/gpt-5.4.toml +++ b/providers/cortecs/models/gpt-5.4.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.4" +# Cortecs Chat maps `reasoning_effort = low|medium|high`; no budget is exposed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) base_model_omit = ["limit.input"] reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/cortecs/models/gpt-oss-120b.toml b/providers/cortecs/models/gpt-oss-120b.toml index 2d90a8edf..b7ebece28 100644 --- a/providers/cortecs/models/gpt-oss-120b.toml +++ b/providers/cortecs/models/gpt-oss-120b.toml @@ -1,4 +1,6 @@ name = "GPT Oss 120b" +# Cortecs Chat maps `reasoning_effort = low|medium|high`; no budget is exposed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "gpt-oss" release_date = "2025-08-05" last_updated = "2025-08-05" diff --git a/providers/cortecs/models/kimi-k2.5.toml b/providers/cortecs/models/kimi-k2.5.toml index b43146c8f..db296e88f 100644 --- a/providers/cortecs/models/kimi-k2.5.toml +++ b/providers/cortecs/models/kimi-k2.5.toml @@ -1,4 +1,6 @@ name = "Kimi K2.5" +# Cortecs maps `thinking.type = enabled|disabled`; no effort or budget is listed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "kimi-thinking" release_date = "2026-01-27" last_updated = "2026-01-27" diff --git a/providers/cortecs/models/kimi-k2.6.toml b/providers/cortecs/models/kimi-k2.6.toml index 53306891d..57e7ab8d3 100644 --- a/providers/cortecs/models/kimi-k2.6.toml +++ b/providers/cortecs/models/kimi-k2.6.toml @@ -1,4 +1,6 @@ name = "Kimi K2.6" +# Cortecs maps `thinking.type = enabled|disabled`; no effort or budget is listed. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) family = "kimi-thinking" release_date = "2026-04-17" last_updated = "2026-04-17" diff --git a/providers/cortecs/provider.toml b/providers/cortecs/provider.toml index 260753a92..6d2607022 100644 --- a/providers/cortecs/provider.toml +++ b/providers/cortecs/provider.toml @@ -1,5 +1,11 @@ name = "Cortecs" env = ["CORTECS_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP is POST `/v1/chat/completions`. Cortecs maps OpenAI +# `reasoning_effort` to backend controls: Claude `output_config.effort` and +# `thinking.budget_tokens`, Kimi `thinking.type`, and GPT/DeepSeek effort. +# Compatibility fields unsupported by a selected backend can be silently +# ignored, so only model-specific controls cited in model comments are usable. +# https://api.cortecs.ai/v1/models (accessed 2026-06-25) api = "https://api.cortecs.ai/v1" doc = "https://api.cortecs.ai/v1/models" \ No newline at end of file diff --git a/providers/crof/models/glm-4.7-flash.toml b/providers/crof/models/glm-4.7-flash.toml index 0239c315e..af616d973 100644 --- a/providers/crof/models/glm-4.7-flash.toml +++ b/providers/crof/models/glm-4.7-flash.toml @@ -1,4 +1,6 @@ base_model = "zhipuai/glm-4.7-flash" +# Crof's request docs establish no model-specific toggle, effort, or budget. +# https://crof.ai/docs.md (accessed 2026-06-25) reasoning_options = [] [cost] diff --git a/providers/crof/models/glm-4.7.toml b/providers/crof/models/glm-4.7.toml index 01646defc..6a0a85919 100644 --- a/providers/crof/models/glm-4.7.toml +++ b/providers/crof/models/glm-4.7.toml @@ -1,4 +1,6 @@ base_model = "zhipuai/glm-4.7" +# Crof's request docs establish no model-specific toggle, effort, or budget. +# https://crof.ai/docs.md (accessed 2026-06-25) reasoning_options = [] [interleaved] diff --git a/providers/crof/models/glm-5.toml b/providers/crof/models/glm-5.toml index 28991c3b3..43a61d989 100644 --- a/providers/crof/models/glm-5.toml +++ b/providers/crof/models/glm-5.toml @@ -1,4 +1,6 @@ base_model = "zhipuai/glm-5" +# Crof's request docs establish no model-specific toggle, effort, or budget. +# https://crof.ai/docs.md (accessed 2026-06-25) reasoning_options = [] [interleaved] diff --git a/providers/crof/models/minimax-m2.5.toml b/providers/crof/models/minimax-m2.5.toml index d95f15721..97a8de0c3 100644 --- a/providers/crof/models/minimax-m2.5.toml +++ b/providers/crof/models/minimax-m2.5.toml @@ -1,4 +1,6 @@ base_model = "minimax/MiniMax-M2.5" +# Crof's request docs establish no model-specific toggle, effort, or budget. +# https://crof.ai/docs.md (accessed 2026-06-25) reasoning = false reasoning_options = [] diff --git a/providers/crof/provider.toml b/providers/crof/provider.toml index e98511587..d83efc9ce 100644 --- a/providers/crof/provider.toml +++ b/providers/crof/provider.toml @@ -1,5 +1,10 @@ name = "CrofAI" env = ["CROF_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw OpenAI Chat is POST `/v1/chat/completions` (also `/v2/chat/completions`); +# Anthropic compatibility is POST `https://anthropic.nahcrof.com/v1/messages`. +# Chat accepts `reasoning_effort = none|low|medium|high`; `none` disables +# reasoning. No reasoning-token budget field is documented. +# https://crof.ai/docs.md (accessed 2026-06-25) api = "https://crof.ai/v1" doc = "https://crof.ai/docs" \ No newline at end of file diff --git a/providers/databricks/models/databricks-claude-haiku-4-5.toml b/providers/databricks/models/databricks-claude-haiku-4-5.toml index 57d8d4b85..adc0dd2f5 100644 --- a/providers/databricks/models/databricks-claude-haiku-4-5.toml +++ b/providers/databricks/models/databricks-claude-haiku-4-5.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-haiku-4-5" +# Databricks' reasoning-model request table documents no toggle, effort, or +# token-budget control for this endpoint. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [] [cost] diff --git a/providers/databricks/models/databricks-claude-opus-4-1.toml b/providers/databricks/models/databricks-claude-opus-4-1.toml index 9d5263f38..f29d56194 100644 --- a/providers/databricks/models/databricks-claude-opus-4-1.toml +++ b/providers/databricks/models/databricks-claude-opus-4-1.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-opus-4-1" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-claude-opus-4-5.toml b/providers/databricks/models/databricks-claude-opus-4-5.toml index ab6bf4feb..634ecf31b 100644 --- a/providers/databricks/models/databricks-claude-opus-4-5.toml +++ b/providers/databricks/models/databricks-claude-opus-4-5.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-opus-4-5" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-claude-opus-4-6.toml b/providers/databricks/models/databricks-claude-opus-4-6.toml index a3884e7d6..972bc2af7 100644 --- a/providers/databricks/models/databricks-claude-opus-4-6.toml +++ b/providers/databricks/models/databricks-claude-opus-4-6.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-opus-4-6" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-claude-opus-4-7.toml b/providers/databricks/models/databricks-claude-opus-4-7.toml index 8468398ea..b1b9d3948 100644 --- a/providers/databricks/models/databricks-claude-opus-4-7.toml +++ b/providers/databricks/models/databricks-claude-opus-4-7.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-opus-4-7" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-claude-sonnet-4-5.toml b/providers/databricks/models/databricks-claude-sonnet-4-5.toml index d747bb8db..aca9af0f6 100644 --- a/providers/databricks/models/databricks-claude-sonnet-4-5.toml +++ b/providers/databricks/models/databricks-claude-sonnet-4-5.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-sonnet-4-5" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-claude-sonnet-4-6.toml b/providers/databricks/models/databricks-claude-sonnet-4-6.toml index ba4c03ce7..134c5cf2c 100644 --- a/providers/databricks/models/databricks-claude-sonnet-4-6.toml +++ b/providers/databricks/models/databricks-claude-sonnet-4-6.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-sonnet-4-6" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-claude-sonnet-4.toml b/providers/databricks/models/databricks-claude-sonnet-4.toml index 2335c64d9..de16463c8 100644 --- a/providers/databricks/models/databricks-claude-sonnet-4.toml +++ b/providers/databricks/models/databricks-claude-sonnet-4.toml @@ -1,4 +1,7 @@ base_model = "anthropic/claude-sonnet-4-5-20250929" +# Chat uses `thinking = { type = "enabled", budget_tokens = N }`; N >= 1024 +# and must be less than `max_tokens`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 1_024 }] [cost] diff --git a/providers/databricks/models/databricks-gemini-2-5-flash.toml b/providers/databricks/models/databricks-gemini-2-5-flash.toml index 22060a7de..b037dfa9b 100644 --- a/providers/databricks/models/databricks-gemini-2-5-flash.toml +++ b/providers/databricks/models/databricks-gemini-2-5-flash.toml @@ -1,4 +1,7 @@ base_model = "google/gemini-2.5-flash" +# Chat uses `thinking.budget_tokens` 0..24576; 0 disables and -1 requests dynamic +# thinking in native Gemini, but -1 is a sentinel rather than a numeric budget. +# https://ai.google.dev/gemini-api/docs/thinking (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 0, max = 24_576 }] [cost] diff --git a/providers/databricks/models/databricks-gemini-2-5-pro.toml b/providers/databricks/models/databricks-gemini-2-5-pro.toml index d5848f8ab..9002bb8de 100644 --- a/providers/databricks/models/databricks-gemini-2-5-pro.toml +++ b/providers/databricks/models/databricks-gemini-2-5-pro.toml @@ -1,4 +1,7 @@ base_model = "google/gemini-2.5-pro" +# Chat uses `thinking.budget_tokens` 128..32768; reasoning cannot be disabled; +# native Gemini -1 requests dynamic thinking but is not a numeric budget. +# https://ai.google.dev/gemini-api/docs/thinking (accessed 2026-06-25) reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] [cost] diff --git a/providers/databricks/models/databricks-gemini-3-1-flash-lite.toml b/providers/databricks/models/databricks-gemini-3-1-flash-lite.toml index 2e8165f38..da7d049be 100644 --- a/providers/databricks/models/databricks-gemini-3-1-flash-lite.toml +++ b/providers/databricks/models/databricks-gemini-3-1-flash-lite.toml @@ -1,4 +1,6 @@ base_model = "google/gemini-3.1-flash-lite-preview" +# Chat `reasoning_effort = low|medium|high`; `low` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gemini-3-1-pro.toml b/providers/databricks/models/databricks-gemini-3-1-pro.toml index 6669e542f..72df2310e 100644 --- a/providers/databricks/models/databricks-gemini-3-1-pro.toml +++ b/providers/databricks/models/databricks-gemini-3-1-pro.toml @@ -1,4 +1,6 @@ base_model = "google/gemini-3.1-pro-preview-customtools" +# Chat `reasoning_effort = low|medium|high`; `low` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gemini-3-flash.toml b/providers/databricks/models/databricks-gemini-3-flash.toml index a2fac1cec..3e6d8ccb7 100644 --- a/providers/databricks/models/databricks-gemini-3-flash.toml +++ b/providers/databricks/models/databricks-gemini-3-flash.toml @@ -1,4 +1,6 @@ base_model = "google/gemini-3-flash-preview" +# Chat `reasoning_effort = low|medium|high`; `low` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gemini-3-pro.toml b/providers/databricks/models/databricks-gemini-3-pro.toml index 1484d2faa..48633a5d4 100644 --- a/providers/databricks/models/databricks-gemini-3-pro.toml +++ b/providers/databricks/models/databricks-gemini-3-pro.toml @@ -1,4 +1,6 @@ base_model = "google/gemini-3-pro-preview" +# Chat `reasoning_effort = low|medium|high`; `low` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-1.toml b/providers/databricks/models/databricks-gpt-5-1.toml index 489f83fff..a47c19cf4 100644 --- a/providers/databricks/models/databricks-gpt-5-1.toml +++ b/providers/databricks/models/databricks-gpt-5-1.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.1" +# Chat `reasoning_effort = none|low|medium|high`; `none` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-2.toml b/providers/databricks/models/databricks-gpt-5-2.toml index b2ea2896d..b79c69497 100644 --- a/providers/databricks/models/databricks-gpt-5-2.toml +++ b/providers/databricks/models/databricks-gpt-5-2.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.2" +# Chat `reasoning_effort = none|low|medium|high`; `none` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-4-mini.toml b/providers/databricks/models/databricks-gpt-5-4-mini.toml index b4f784d29..6abf090a6 100644 --- a/providers/databricks/models/databricks-gpt-5-4-mini.toml +++ b/providers/databricks/models/databricks-gpt-5-4-mini.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.4-mini" +# Chat `reasoning_effort = low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-4-nano.toml b/providers/databricks/models/databricks-gpt-5-4-nano.toml index d5470ad17..25b335f9c 100644 --- a/providers/databricks/models/databricks-gpt-5-4-nano.toml +++ b/providers/databricks/models/databricks-gpt-5-4-nano.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.4-nano" +# Chat `reasoning_effort = low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-4.toml b/providers/databricks/models/databricks-gpt-5-4.toml index fa943ffcc..5dd069c37 100644 --- a/providers/databricks/models/databricks-gpt-5-4.toml +++ b/providers/databricks/models/databricks-gpt-5-4.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.4" +# Chat `reasoning_effort = low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-5.toml b/providers/databricks/models/databricks-gpt-5-5.toml index 0871b3e9a..31ad40cc5 100644 --- a/providers/databricks/models/databricks-gpt-5-5.toml +++ b/providers/databricks/models/databricks-gpt-5-5.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5.5" +# Chat `reasoning_effort = low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-mini.toml b/providers/databricks/models/databricks-gpt-5-mini.toml index 5754d47df..b537657fa 100644 --- a/providers/databricks/models/databricks-gpt-5-mini.toml +++ b/providers/databricks/models/databricks-gpt-5-mini.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5-mini" +# Chat `reasoning_effort = minimal|low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5-nano.toml b/providers/databricks/models/databricks-gpt-5-nano.toml index b96eddfd6..7efa8abad 100644 --- a/providers/databricks/models/databricks-gpt-5-nano.toml +++ b/providers/databricks/models/databricks-gpt-5-nano.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5-nano" +# Chat `reasoning_effort = minimal|low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-5.toml b/providers/databricks/models/databricks-gpt-5.toml index 104575502..81dda564d 100644 --- a/providers/databricks/models/databricks-gpt-5.toml +++ b/providers/databricks/models/databricks-gpt-5.toml @@ -1,4 +1,6 @@ base_model = "openai/gpt-5" +# Chat `reasoning_effort = minimal|low|medium|high`; Responses uses `reasoning.effort`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] [cost] diff --git a/providers/databricks/models/databricks-gpt-oss-120b.toml b/providers/databricks/models/databricks-gpt-oss-120b.toml index bb35cbd96..2f4dc9d7b 100644 --- a/providers/databricks/models/databricks-gpt-oss-120b.toml +++ b/providers/databricks/models/databricks-gpt-oss-120b.toml @@ -4,6 +4,8 @@ release_date = "2025-08-05" last_updated = "2025-08-05" attachment = false reasoning = true +# Chat `reasoning_effort = low|medium|high`; `medium` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/databricks/models/databricks-gpt-oss-20b.toml b/providers/databricks/models/databricks-gpt-oss-20b.toml index 0fbed0dd9..f04bb3522 100644 --- a/providers/databricks/models/databricks-gpt-oss-20b.toml +++ b/providers/databricks/models/databricks-gpt-oss-20b.toml @@ -4,6 +4,8 @@ release_date = "2025-08-05" last_updated = "2025-08-05" attachment = false reasoning = true +# Chat `reasoning_effort = low|medium|high`; `medium` is the default. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/databricks/provider.toml b/providers/databricks/provider.toml index f07eaf4ab..10e68f232 100644 --- a/providers/databricks/provider.toml +++ b/providers/databricks/provider.toml @@ -1,5 +1,15 @@ name = "Databricks" npm = "@ai-sdk/openai-compatible" +# Raw Chat is POST `/serving-endpoints/{model}/invocations`: GPT/Gemini 3 use +# `reasoning_effort`; Claude/Gemini 2.5 use +# `thinking = { type = "enabled", budget_tokens = N }`. +# Native OpenAI Responses is POST `/serving-endpoints/responses` and uses +# `reasoning.effort`; Open Responses supports Claude, Gemini, and open models. +# Native Gemini `generateContent` uses `generationConfig.thinkingConfig`. +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-reason-models (accessed 2026-06-25) +# https://docs.databricks.com/aws/en/machine-learning/foundation-model-apis/api-reference (accessed 2026-06-25) +# https://docs.databricks.com/aws/en/machine-learning/model-serving/query-open-responses-models (accessed 2026-06-25) +# https://docs.databricks.com/aws/en/generative-ai/external-models/google-gemini (accessed 2026-06-25) api = "https://${DATABRICKS_HOST}/ai-gateway/mlflow/v1" env = ["DATABRICKS_HOST", "DATABRICKS_TOKEN"] doc = "https://docs.databricks.com/aws/en/machine-learning/foundation-models/" diff --git a/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Flash.toml b/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Flash.toml index 6c708c932..504127b10 100644 --- a/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Flash.toml +++ b/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Flash.toml @@ -1,4 +1,9 @@ base_model = "deepseek/deepseek-v4-flash" +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# reasoning_effort = "low"|"medium"|"high"|"xhigh"; "none" disables reasoning. +# Sources: +# https://docs.deepinfra.com/api-reference/chat-completions/openai-chat-completions +# https://deepinfra.com/deepseek-ai/DeepSeek-V4-Flash/api reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh"] }] [interleaved] diff --git a/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Pro.toml b/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Pro.toml index fb50a1e0f..5c5e98f99 100644 --- a/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Pro.toml +++ b/providers/deepinfra/models/deepseek-ai/DeepSeek-V4-Pro.toml @@ -1,4 +1,9 @@ base_model = "deepseek/deepseek-v4-pro" +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# reasoning_effort = "low"|"medium"|"high"|"xhigh"; "none" disables reasoning. +# Sources: +# https://docs.deepinfra.com/api-reference/chat-completions/openai-chat-completions +# https://deepinfra.com/deepseek-ai/DeepSeek-V4-Pro/api reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh"] }] [interleaved] diff --git a/providers/deepinfra/models/openai/gpt-oss-120b.toml b/providers/deepinfra/models/openai/gpt-oss-120b.toml index 734ae2d0f..f4eb4a86b 100644 --- a/providers/deepinfra/models/openai/gpt-oss-120b.toml +++ b/providers/deepinfra/models/openai/gpt-oss-120b.toml @@ -1,4 +1,9 @@ # https://deepinfra.com/openai/gpt-oss-120b +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# reasoning_effort = "low"|"medium"|"high". +# Sources: +# https://docs.deepinfra.com/api-reference/chat-completions/openai-chat-completions +# https://deepinfra.com/openai/gpt-oss-120b/api name = "GPT OSS 120B" family = "gpt-oss" diff --git a/providers/deepinfra/models/openai/gpt-oss-20b.toml b/providers/deepinfra/models/openai/gpt-oss-20b.toml index eebce8f2e..3ef0442eb 100644 --- a/providers/deepinfra/models/openai/gpt-oss-20b.toml +++ b/providers/deepinfra/models/openai/gpt-oss-20b.toml @@ -1,4 +1,9 @@ # https://deepinfra.com/openai/gpt-oss-20b +# Model-specific reasoning HTTP values (accessed 2026-06-25): +# reasoning_effort = "low"|"medium"|"high". +# Sources: +# https://docs.deepinfra.com/api-reference/chat-completions/openai-chat-completions +# https://deepinfra.com/openai/gpt-oss-20b/api name = "GPT OSS 20B" family = "gpt-oss" diff --git a/providers/deepinfra/provider.toml b/providers/deepinfra/provider.toml index f1ffa7a37..b0ce18920 100644 --- a/providers/deepinfra/provider.toml +++ b/providers/deepinfra/provider.toml @@ -1,4 +1,18 @@ name = "Deep Infra" env = ["DEEPINFRA_API_KEY"] npm = "@ai-sdk/deepinfra" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI Chat: POST https://api.deepinfra.com/v1/openai/chat/completions +# (also POST /v1/chat/completions). Toggle: reasoning.enabled = true|false; +# reasoning_effort = "none" also disables. Effort: reasoning_effort or +# reasoning.effort = "low"|"medium"|"high"|"xhigh". Token budget: not documented. +# Anthropic Messages: POST https://api.deepinfra.com/anthropic/v1/messages; +# toggle: thinking.enabled = true|false; effort and token budget: not documented. +# Native: POST https://api.deepinfra.com/v1/inference/{model_name}; request fields +# are model-schema-specific, with no shared reasoning control documented. +# Sources: +# https://docs.deepinfra.com/chat/reasoning +# https://docs.deepinfra.com/api-reference/chat-completions/openai-chat-completions +# https://docs.deepinfra.com/api-reference/chat-completions/anthropic-messages +# https://docs.deepinfra.com/apis/deepinfra-native doc = "https://deepinfra.com/models" diff --git a/providers/deepseek/models/deepseek-reasoner.toml b/providers/deepseek/models/deepseek-reasoner.toml index 4383ea02f..af7d26c8c 100644 --- a/providers/deepseek/models/deepseek-reasoner.toml +++ b/providers/deepseek/models/deepseek-reasoner.toml @@ -1,4 +1,7 @@ name = "DeepSeek Reasoner" +# Legacy reasoning-only request schema documents no toggle, effort, or token +# budget; reasoning is returned as `reasoning_content`. +# https://api-docs.deepseek.com/guides/reasoning_model (accessed 2026-06-25) family = "deepseek-thinking" release_date = "2025-12-01" last_updated = "2026-02-28" diff --git a/providers/deepseek/models/deepseek-v4-flash.toml b/providers/deepseek/models/deepseek-v4-flash.toml index dccb6d24b..aa74fa1b0 100644 --- a/providers/deepseek/models/deepseek-v4-flash.toml +++ b/providers/deepseek/models/deepseek-v4-flash.toml @@ -1,4 +1,7 @@ name = "DeepSeek V4 Flash" +# OpenAI: `thinking.type = enabled|disabled`, `reasoning_effort = high|max`. +# Anthropic: `thinking.type`, `output_config.effort = high|max`; budget ignored. +# https://api-docs.deepseek.com/api/create-chat-completion (accessed 2026-06-25) family = "deepseek-flash" release_date = "2026-04-24" last_updated = "2026-04-24" diff --git a/providers/deepseek/models/deepseek-v4-pro.toml b/providers/deepseek/models/deepseek-v4-pro.toml index deca5e7c8..45ca60c8e 100644 --- a/providers/deepseek/models/deepseek-v4-pro.toml +++ b/providers/deepseek/models/deepseek-v4-pro.toml @@ -1,4 +1,7 @@ name = "DeepSeek V4 Pro" +# OpenAI: `thinking.type = enabled|disabled`, `reasoning_effort = high|max`. +# Anthropic: `thinking.type`, `output_config.effort = high|max`; budget ignored. +# https://api-docs.deepseek.com/api/create-chat-completion (accessed 2026-06-25) family = "deepseek-thinking" release_date = "2026-04-24" last_updated = "2026-04-24" diff --git a/providers/deepseek/provider.toml b/providers/deepseek/provider.toml index bceb3b7f2..115adcf7e 100644 --- a/providers/deepseek/provider.toml +++ b/providers/deepseek/provider.toml @@ -1,5 +1,11 @@ name = "DeepSeek" env = ["DEEPSEEK_API_KEY"] npm = "@ai-sdk/openai-compatible" +# OpenAI Chat is POST `/chat/completions`: `thinking.type = enabled|disabled` +# and `reasoning_effort = high|max`; low/medium map to high and xhigh to max. +# Anthropic Messages is POST `/anthropic/v1/messages`: `thinking.type` and +# `output_config.effort = high|max`; `thinking.budget_tokens` is ignored. +# https://api-docs.deepseek.com/guides/thinking_mode (accessed 2026-06-25) +# https://api-docs.deepseek.com/guides/anthropic_api (accessed 2026-06-25) doc = "https://api-docs.deepseek.com/quick_start/pricing" api = "https://api.deepseek.com" diff --git a/providers/digitalocean/models/openai-gpt-5.4-mini.toml b/providers/digitalocean/models/openai-gpt-5.4-mini.toml index e4cfe2d06..929e221fd 100644 --- a/providers/digitalocean/models/openai-gpt-5.4-mini.toml +++ b/providers/digitalocean/models/openai-gpt-5.4-mini.toml @@ -1,4 +1,8 @@ name = "GPT-5.4 mini" +# Serverless supports only POST `/v1/responses`; use +# `reasoning.effort = none|low|medium|high|xhigh`. +# https://docs.digitalocean.com/products/inference/details/models/index.html.md (accessed 2026-06-25) +# https://developers.openai.com/api/docs/models/gpt-5.4-mini (accessed 2026-06-25) family = "gpt-mini" release_date = "2026-03-17" last_updated = "2026-03-17" diff --git a/providers/digitalocean/models/openai-gpt-5.4-nano.toml b/providers/digitalocean/models/openai-gpt-5.4-nano.toml index dd7380fa1..bf528ee01 100644 --- a/providers/digitalocean/models/openai-gpt-5.4-nano.toml +++ b/providers/digitalocean/models/openai-gpt-5.4-nano.toml @@ -1,4 +1,8 @@ name = "GPT-5.4 nano" +# Serverless supports only POST `/v1/responses`; use +# `reasoning.effort = none|low|medium|high|xhigh`. +# https://docs.digitalocean.com/products/inference/details/models/index.html.md (accessed 2026-06-25) +# https://developers.openai.com/api/docs/models/gpt-5.4-nano (accessed 2026-06-25) family = "gpt-nano" release_date = "2026-03-17" last_updated = "2026-03-17" diff --git a/providers/digitalocean/models/openai-gpt-5.4-pro.toml b/providers/digitalocean/models/openai-gpt-5.4-pro.toml index 6cf3ba182..58b406394 100644 --- a/providers/digitalocean/models/openai-gpt-5.4-pro.toml +++ b/providers/digitalocean/models/openai-gpt-5.4-pro.toml @@ -1,4 +1,8 @@ name = "GPT-5.4 pro" +# Serverless supports only POST `/v1/responses`; this model restricts +# `reasoning.effort` to medium|high|xhigh. +# https://docs.digitalocean.com/products/inference/details/models/index.html.md (accessed 2026-06-25) +# https://developers.openai.com/api/docs/models/gpt-5.4-pro (accessed 2026-06-25) family = "gpt-pro" release_date = "2026-03-05" last_updated = "2026-03-05" diff --git a/providers/digitalocean/models/openai-gpt-5.4.toml b/providers/digitalocean/models/openai-gpt-5.4.toml index b57293e3d..32707694f 100644 --- a/providers/digitalocean/models/openai-gpt-5.4.toml +++ b/providers/digitalocean/models/openai-gpt-5.4.toml @@ -1,4 +1,8 @@ name = "GPT-5.4" +# Serverless supports only POST `/v1/responses`; use +# `reasoning.effort = none|low|medium|high|xhigh`. +# https://docs.digitalocean.com/products/inference/details/models/index.html.md (accessed 2026-06-25) +# https://developers.openai.com/api/docs/models/gpt-5.4 (accessed 2026-06-25) family = "gpt" release_date = "2026-03-05" last_updated = "2026-03-05" diff --git a/providers/digitalocean/models/openai-gpt-5.5.toml b/providers/digitalocean/models/openai-gpt-5.5.toml index 1877f29b4..b4278fd89 100644 --- a/providers/digitalocean/models/openai-gpt-5.5.toml +++ b/providers/digitalocean/models/openai-gpt-5.5.toml @@ -1,4 +1,8 @@ name = "GPT-5.5" +# Serverless supports only POST `/v1/responses`; use +# `reasoning.effort = none|low|medium|high|xhigh`. +# https://docs.digitalocean.com/products/inference/details/models/index.html.md (accessed 2026-06-25) +# https://developers.openai.com/api/docs/models/gpt-5.5 (accessed 2026-06-25) family = "gpt" release_date = "2026-04-23" last_updated = "2026-04-30" diff --git a/providers/digitalocean/provider.toml b/providers/digitalocean/provider.toml index 852e14c91..8b30eef3a 100644 --- a/providers/digitalocean/provider.toml +++ b/providers/digitalocean/provider.toml @@ -1,5 +1,12 @@ name = "DigitalOcean" env = ["DIGITALOCEAN_ACCESS_TOKEN"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP routes are POST `/v1/chat/completions`, `/v1/responses`, and +# `/v1/messages`. Chat/Responses accept OpenAI `reasoning_effort` or +# `reasoning.effort = none|low|medium|high|max`; `reasoning.max_tokens` is an +# Anthropic reasoning budget, otherwise derived as 20/50/80/95 percent of +# `max_completion_tokens` for low/medium/high/max. +# https://docs.digitalocean.com/products/inference/how-to/si-endpoints/index.html.md (accessed 2026-06-25) +# https://docs.digitalocean.com/products/inference/how-to/use-reasoning/index.html.md (accessed 2026-06-25) api = "https://inference.do-ai.run/v1" doc = "https://docs.digitalocean.com/products/gradient-ai-platform/details/models/" diff --git a/providers/dinference/provider.toml b/providers/dinference/provider.toml index c22810a2d..0b3775cad 100644 --- a/providers/dinference/provider.toml +++ b/providers/dinference/provider.toml @@ -1,5 +1,8 @@ name = "DInference" env = ["DINFERENCE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP is POST `/v1/chat/completions`. The published request documentation +# lists no reasoning toggle, effort, or token-budget field. +# https://dinference.com/docs (accessed 2026-06-25) doc = "https://dinference.com" api = "https://api.dinference.com/v1" diff --git a/providers/drun/provider.toml b/providers/drun/provider.toml index d51044cde..ca048544e 100644 --- a/providers/drun/provider.toml +++ b/providers/drun/provider.toml @@ -1,5 +1,8 @@ name = "D.Run (China)" npm = "@ai-sdk/openai-compatible" +# Raw HTTP is POST `/v1/chat/completions`. The public d.run documentation and +# request surface document no reasoning toggle, effort, or token-budget field. +# https://www.d.run (accessed 2026-06-25) env = ["DRUN_API_KEY"] api = "https://chat.d.run/v1" doc = "https://www.d.run" diff --git a/providers/evroc/provider.toml b/providers/evroc/provider.toml index 3d64e64c4..ae92330c1 100644 --- a/providers/evroc/provider.toml +++ b/providers/evroc/provider.toml @@ -1,4 +1,8 @@ name = "evroc" +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/chat/completions. +# evroc documents OpenAI compatibility but no toggle, effort, or budget field; +# the raw reasoning-control surface is undocumented. +# https://docs.evroc.com/products/think/overview.html env = ["EVROC_API_KEY"] npm = "@ai-sdk/openai-compatible" doc = "https://docs.evroc.com/products/think/overview.html" diff --git a/providers/fastrouter/provider.toml b/providers/fastrouter/provider.toml index ffa8b480d..19f6a58cf 100644 --- a/providers/fastrouter/provider.toml +++ b/providers/fastrouter/provider.toml @@ -1,4 +1,9 @@ name = "FastRouter" +# Reasoning HTTP format (accessed 2026-06-25): POST /api/v1/chat/completions. +# Controls are route-specific; use live GET /api/v1/models metadata and each +# route's mappings rather than assuming an upstream field is passed through. +# https://go.fastrouter.ai/api/v1/models +# https://fastrouter.ai/models env = ["FASTROUTER_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://go.fastrouter.ai/api/v1" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-flash.toml b/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-flash.toml index 3dcbe4523..21fab17a9 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-flash.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-flash.toml @@ -1,4 +1,6 @@ base_model = "deepseek/deepseek-v4-flash" +# Fireworks' published schema does not document this model's toggle or "max" +# effort mapping. https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) last_updated = "2026-06-16" [[reasoning_options]] diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-pro.toml b/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-pro.toml index 32e2e20e7..eda5cdf4d 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-pro.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/deepseek-v4-pro.toml @@ -1,4 +1,6 @@ base_model = "deepseek/deepseek-v4-pro" +# Fireworks' published schema does not document this model's toggle or "max" +# effort mapping. https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) [[reasoning_options]] type = "toggle" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p1.toml b/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p1.toml index 9a312633c..c0bbbfa51 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p1.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p1.toml @@ -1,4 +1,6 @@ name = "GLM 5.1" +# Fireworks documents thinking.type = "enabled", but no explicit disabled +# value for this model. https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "glm" release_date = "2026-04-01" last_updated = "2026-04-01" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p2.toml b/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p2.toml index 30062c3c7..c367b2487 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p2.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/glm-5p2.toml @@ -1,4 +1,6 @@ name = "GLM 5.2" +# Fireworks' published schema does not document this model's toggle or "max" +# effort mapping. https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "glm" release_date = "2026-06-16" last_updated = "2026-06-16" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-120b.toml b/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-120b.toml index 7e192b243..467c3e40a 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-120b.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-120b.toml @@ -1,4 +1,6 @@ name = "GPT OSS 120B" +# $.reasoning_effort = "low" | "medium" | "high". +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "gpt-oss" release_date = "2025-08-05" last_updated = "2026-06-16" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-20b.toml b/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-20b.toml index afcc2f8a8..ab2903722 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-20b.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/gpt-oss-20b.toml @@ -1,4 +1,6 @@ name = "GPT OSS 20B" +# $.reasoning_effort = "low" | "medium" | "high". +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "gpt-oss" release_date = "2025-08-05" last_updated = "2025-08-05" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p6.toml b/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p6.toml index b0f3d1737..66b0b8610 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p6.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p6.toml @@ -1,4 +1,6 @@ name = "Kimi K2.6" +# Fireworks documents thinking.type = "enabled", but no explicit disabled +# value for this model. https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "kimi-thinking" release_date = "2026-04-17" last_updated = "2026-04-17" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p7-code.toml b/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p7-code.toml index 954721c58..6f839439a 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p7-code.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/kimi-k2p7-code.toml @@ -1,4 +1,6 @@ name = "Kimi K2.7 Code" +# Fireworks documents thinking.type = "enabled", but no explicit disabled +# value for this model. https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "kimi-k2" release_date = "2026-06-12" last_updated = "2026-06-16" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m2p7.toml b/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m2p7.toml index af07146a4..ed11119f3 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m2p7.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m2p7.toml @@ -1,4 +1,6 @@ name = "MiniMax-M2.7" +# $.reasoning_effort = "low" | "medium" | "high". +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "minimax" release_date = "2026-04-12" last_updated = "2026-04-12" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m3.toml b/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m3.toml index 2732942f3..5c650fce0 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m3.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/minimax-m3.toml @@ -1,4 +1,6 @@ name = "MiniMax-M3" +# $.reasoning_effort = "low" | "medium" | "high". +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "minimax" release_date = "2026-06-12" last_updated = "2026-06-12" diff --git a/providers/fireworks-ai/models/accounts/fireworks/models/qwen3p7-plus.toml b/providers/fireworks-ai/models/accounts/fireworks/models/qwen3p7-plus.toml index 55a6010eb..8ffa141d6 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/models/qwen3p7-plus.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/models/qwen3p7-plus.toml @@ -1,4 +1,7 @@ name = "Qwen 3.7 Plus" +# Fireworks documents effort low|medium|high and thinking.budget_tokens >= 1024; +# toggle semantics and configured min 1 are undocumented. +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "qwen" release_date = "2026-06-12" last_updated = "2026-06-12" diff --git a/providers/fireworks-ai/models/accounts/fireworks/routers/glm-5p1-fast.toml b/providers/fireworks-ai/models/accounts/fireworks/routers/glm-5p1-fast.toml index 04aa0a1f7..f194afa3b 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/routers/glm-5p1-fast.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/routers/glm-5p1-fast.toml @@ -1,4 +1,6 @@ name = "GLM 5.1 Fast" +# No router-specific toggle mapping or explicit disabled value is documented. +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "glm" release_date = "2026-04-01" last_updated = "2026-04-01" diff --git a/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-fast.toml b/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-fast.toml index f961532fb..77ce7c0af 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-fast.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-fast.toml @@ -1,4 +1,6 @@ name = "Kimi K2.6 Fast" +# No router-specific toggle mapping or explicit disabled value is documented. +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "kimi-thinking" release_date = "2026-04-17" last_updated = "2026-06-05" diff --git a/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-turbo.toml b/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-turbo.toml index 597c87833..b4793f5c8 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-turbo.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p6-turbo.toml @@ -1,4 +1,6 @@ name = "Kimi K2.6 Turbo" +# No router-specific toggle mapping or explicit disabled value is documented. +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "kimi-thinking" release_date = "2026-04-17" last_updated = "2026-04-17" diff --git a/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p7-code-fast.toml b/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p7-code-fast.toml index a7d123840..320bbd198 100644 --- a/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p7-code-fast.toml +++ b/providers/fireworks-ai/models/accounts/fireworks/routers/kimi-k2p7-code-fast.toml @@ -1,4 +1,6 @@ name = "Kimi K2.7 Code Fast" +# No router-specific toggle mapping or explicit disabled value is documented. +# https://docs.fireworks.ai/guides/reasoning (accessed 2026-06-25) family = "kimi-k2" release_date = "2026-06-12" last_updated = "2026-06-16" diff --git a/providers/fireworks-ai/provider.toml b/providers/fireworks-ai/provider.toml index 990c96fd3..abb0b6748 100644 --- a/providers/fireworks-ai/provider.toml +++ b/providers/fireworks-ai/provider.toml @@ -1,4 +1,8 @@ name = "Fireworks AI" +# Reasoning HTTP format (accessed 2026-06-25): POST /inference/v1/chat/completions. +# JSON reasoning_effort: "low" | "medium" | "high", or thinking = +# { type = "enabled", budget_tokens = N } with N >= 1024; the two conflict. +# Support is model-specific. https://docs.fireworks.ai/guides/reasoning env = ["FIREWORKS_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.fireworks.ai/inference/v1/" diff --git a/providers/freemodel/provider.toml b/providers/freemodel/provider.toml index b7e9e37df..c1598c009 100644 --- a/providers/freemodel/provider.toml +++ b/providers/freemodel/provider.toml @@ -1,4 +1,8 @@ name = "FreeModel" +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/messages. +# FreeModel publishes no request schema for toggle, effort, or budget controls; +# its public site does not document the raw API surface. +# https://freemodel.dev env = ["FREEMODEL_API_KEY"] npm = "@ai-sdk/anthropic" api = "https://cc.freemodel.dev/v1" diff --git a/providers/friendli/models/zai-org/GLM-5.2.toml b/providers/friendli/models/zai-org/GLM-5.2.toml index 27afe9c9f..81fbbb90c 100644 --- a/providers/friendli/models/zai-org/GLM-5.2.toml +++ b/providers/friendli/models/zai-org/GLM-5.2.toml @@ -1,4 +1,7 @@ name = "GLM-5.2" +# Friendli documents only $.chat_template_kwargs.enable_thinking = true | false +# for this model, not the configured effort values "high" and "max". +# https://friendli.ai/docs/guides/reasoning (accessed 2026-06-25) base_model = "zhipuai/glm-5.2" [[reasoning_options]] diff --git a/providers/friendli/provider.toml b/providers/friendli/provider.toml index 277b34f7b..7b6196470 100644 --- a/providers/friendli/provider.toml +++ b/providers/friendli/provider.toml @@ -1,4 +1,8 @@ name = "Friendli" +# Reasoning HTTP format (accessed 2026-06-25): POST /serverless/v1/chat/completions. +# Toggle: chat_template_kwargs.enable_thinking = true | false; support is +# model-specific. No effort or reasoning-token budget field is documented. +# https://friendli.ai/docs/guides/reasoning env = ["FRIENDLI_TOKEN"] npm = "@ai-sdk/openai-compatible" api = "https://api.friendli.ai/serverless/v1" diff --git a/providers/frogbot/provider.toml b/providers/frogbot/provider.toml index 6b755dfc8..0a83cf690 100644 --- a/providers/frogbot/provider.toml +++ b/providers/frogbot/provider.toml @@ -1,4 +1,11 @@ name = "FrogBot" +# Reasoning HTTP format (accessed 2026-06-25): POST /api/v1/chat/completions. +# JSON reasoning_effort: "low" | "medium" | "high"; Anthropic models also +# accept thinking = { type = "enabled", budget_tokens = N }. +# https://docs.frogbot.ai/api-reference/chat-completions +# Native Claude POST /api/v1/messages documents no thinking field, so the +# Chat control's native mapping and budget bounds are undocumented. +# https://docs.frogbot.ai/api-reference/messages # Token for FrogBot's OpenAI-compatible proxy env = ["FROGBOT_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/github-copilot/provider.toml b/providers/github-copilot/provider.toml index 6ba3a8703..3037bab5d 100644 --- a/providers/github-copilot/provider.toml +++ b/providers/github-copilot/provider.toml @@ -1,4 +1,10 @@ name = "GitHub Copilot" +# Reasoning HTTP format (accessed 2026-06-25): POST /chat/completions. +# Supported controls and values come from authenticated GET /models metadata +# and can change by account/model; GitHub's public model comparison does not +# document this private request schema or static mappings. +# https://api.githubcopilot.com/models +# https://docs.github.com/en/copilot/reference/ai-models/model-comparison env = ["GITHUB_TOKEN"] npm = "@ai-sdk/openai-compatible" doc = "https://docs.github.com/en/copilot" diff --git a/providers/github-models/provider.toml b/providers/github-models/provider.toml index 48c2a4f72..767543405 100644 --- a/providers/github-models/provider.toml +++ b/providers/github-models/provider.toml @@ -1,4 +1,8 @@ name = "GitHub Models" +# Reasoning HTTP format (accessed 2026-06-25): POST /inference/chat/completions. +# GitHub's request schema has no toggle, effort, or budget property; therefore +# the raw reasoning-control format for listed reasoning models is undocumented. +# https://docs.github.com/en/rest/models/inference env = ["GITHUB_TOKEN"] npm = "@ai-sdk/openai-compatible" doc = "https://docs.github.com/en/github-models" diff --git a/providers/gitlab/provider.toml b/providers/gitlab/provider.toml index 037c5bd38..d9c42b94a 100644 --- a/providers/gitlab/provider.toml +++ b/providers/gitlab/provider.toml @@ -1,4 +1,21 @@ name = "GitLab Duo" env = ["GITLAB_TOKEN"] npm = "gitlab-ai-provider" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# - OpenAI Chat: POST /ai/v1/proxy/openai/v1/chat/completions uses top-level +# `reasoning_effort`; Responses: POST /ai/v1/proxy/openai/v1/responses uses +# `reasoning = { effort = "..." }`. +# https://platform.openai.com/docs/api-reference/chat/create#chat-create-reasoning_effort +# https://platform.openai.com/docs/api-reference/responses/create#responses-create-reasoning +# - Anthropic: POST /ai/v1/proxy/anthropic/v1/messages uses +# `thinking = { type = "enabled", budget_tokens = N }` or +# `thinking = { type = "disabled" }`; effort is `output_config.effort`. +# https://docs.anthropic.com/en/api/messages +# The npm provider obtains a direct-access token, points the native SDKs at +# those proxy bases, but builds fixed request bodies without reasoning fields; +# it does not expose a reasoning passthrough. These raw shapes therefore do not +# make `reasoning_options` available through this configured npm integration. +# https://gitlab.com/vglafirov/gitlab-ai-provider/-/blob/main/src/gitlab-direct-access.ts +# https://gitlab.com/vglafirov/gitlab-ai-provider/-/blob/main/src/gitlab-openai-language-model.ts +# https://gitlab.com/vglafirov/gitlab-ai-provider/-/blob/main/src/gitlab-anthropic-language-model.ts doc = "https://docs.gitlab.com/user/duo_agent_platform/" diff --git a/providers/gmicloud/provider.toml b/providers/gmicloud/provider.toml index df9eff8bd..d9bb6f741 100644 --- a/providers/gmicloud/provider.toml +++ b/providers/gmicloud/provider.toml @@ -1,4 +1,8 @@ name = "GMI Cloud" +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/messages for Claude. +# Native Messages uses thinking = { type = "enabled", budget_tokens = N }; +# GMI examples do not document disable, effort values, or numeric bounds. +# https://docs.gmicloud.ai/model-quickstarts/text/anthropic-claude-opus-4-6 env = ["GMICLOUD_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.gmi-serving.com/v1" diff --git a/providers/google-vertex/models/deepseek-ai/deepseek-v3.1-maas.toml b/providers/google-vertex/models/deepseek-ai/deepseek-v3.1-maas.toml index 6891a6850..d73ee26d0 100644 --- a/providers/google-vertex/models/deepseek-ai/deepseek-v3.1-maas.toml +++ b/providers/google-vertex/models/deepseek-ai/deepseek-v3.1-maas.toml @@ -1,4 +1,7 @@ name = "DeepSeek V3.1" +# POST .../endpoints/openapi/chat/completions; Vertex documents no toggle, +# effort, or reasoning-token budget control for this model. +# https://cloud.google.com/vertex-ai/generative-ai/docs/maas/capabilities/thinking (accessed 2026-06-25) family = "deepseek" release_date = "2025-08-28" last_updated = "2025-08-28" diff --git a/providers/google-vertex/models/deepseek-ai/deepseek-v3.2-maas.toml b/providers/google-vertex/models/deepseek-ai/deepseek-v3.2-maas.toml index 5ac52ac69..77338d419 100644 --- a/providers/google-vertex/models/deepseek-ai/deepseek-v3.2-maas.toml +++ b/providers/google-vertex/models/deepseek-ai/deepseek-v3.2-maas.toml @@ -1,4 +1,7 @@ name = "DeepSeek V3.2" +# POST .../endpoints/openapi/chat/completions; Vertex documents no toggle, +# effort, or reasoning-token budget control for this model. +# https://cloud.google.com/vertex-ai/generative-ai/docs/maas/capabilities/thinking (accessed 2026-06-25) family = "deepseek" release_date = "2025-12-17" last_updated = "2026-04-04" diff --git a/providers/google-vertex/models/meta/llama-3.3-70b-instruct-maas.toml b/providers/google-vertex/models/meta/llama-3.3-70b-instruct-maas.toml index 5b917fbc5..45d106228 100644 --- a/providers/google-vertex/models/meta/llama-3.3-70b-instruct-maas.toml +++ b/providers/google-vertex/models/meta/llama-3.3-70b-instruct-maas.toml @@ -1,4 +1,7 @@ name = "Llama 3.3 70B Instruct" +# POST .../endpoints/openapi/chat/completions with model, messages, max_tokens, +# stream, and extra_body.google.model_safety_settings; no reasoning control documented. +# https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/llama/use-llama (accessed 2026-06-25) family = "llama" release_date = "2025-04-29" last_updated = "2025-04-29" diff --git a/providers/google-vertex/models/meta/llama-4-maverick-17b-128e-instruct-maas.toml b/providers/google-vertex/models/meta/llama-4-maverick-17b-128e-instruct-maas.toml index 4e427fc8b..f36e15579 100644 --- a/providers/google-vertex/models/meta/llama-4-maverick-17b-128e-instruct-maas.toml +++ b/providers/google-vertex/models/meta/llama-4-maverick-17b-128e-instruct-maas.toml @@ -1,4 +1,7 @@ name = "Llama 4 Maverick 17B 128E Instruct" +# POST .../endpoints/openapi/chat/completions with model, messages, max_tokens, +# stream, and extra_body.google.model_safety_settings; no reasoning control documented. +# https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/llama/use-llama (accessed 2026-06-25) family = "llama" release_date = "2025-04-29" last_updated = "2025-04-29" diff --git a/providers/google-vertex/models/moonshotai/kimi-k2-thinking-maas.toml b/providers/google-vertex/models/moonshotai/kimi-k2-thinking-maas.toml index 1cc3ed5fa..515670301 100644 --- a/providers/google-vertex/models/moonshotai/kimi-k2-thinking-maas.toml +++ b/providers/google-vertex/models/moonshotai/kimi-k2-thinking-maas.toml @@ -1,4 +1,7 @@ name = "Kimi K2 Thinking" +# POST .../endpoints/openapi/chat/completions; this is always-thinking, and +# Vertex documents no toggle, effort, or budget request field for it. +# https://cloud.google.com/vertex-ai/generative-ai/docs/maas/capabilities/thinking (accessed 2026-06-25) family = "kimi-thinking" release_date = "2025-11-13" last_updated = "2025-11-13" diff --git a/providers/google-vertex/models/qwen/qwen3-235b-a22b-instruct-2507-maas.toml b/providers/google-vertex/models/qwen/qwen3-235b-a22b-instruct-2507-maas.toml index 9501b8e17..47a2d0f7a 100644 --- a/providers/google-vertex/models/qwen/qwen3-235b-a22b-instruct-2507-maas.toml +++ b/providers/google-vertex/models/qwen/qwen3-235b-a22b-instruct-2507-maas.toml @@ -1,4 +1,7 @@ name = "Qwen3 235B A22B Instruct" +# POST .../endpoints/openapi/chat/completions; this Instruct variant has no +# documented toggle, effort, or reasoning-token budget field. +# https://cloud.google.com/vertex-ai/generative-ai/docs/maas/capabilities/thinking (accessed 2026-06-25) family = "qwen" release_date = "2025-08-13" last_updated = "2025-08-13" diff --git a/providers/google-vertex/models/zai-org/glm-4.7-maas.toml b/providers/google-vertex/models/zai-org/glm-4.7-maas.toml index 36a9f3b9a..ef8417d7b 100644 --- a/providers/google-vertex/models/zai-org/glm-4.7-maas.toml +++ b/providers/google-vertex/models/zai-org/glm-4.7-maas.toml @@ -1,4 +1,7 @@ name = "GLM-4.7" +# POST .../endpoints/openapi/chat/completions; toggle with +# $.chat_template_kwargs.enable_thinking = true | false. +# https://cloud.google.com/vertex-ai/generative-ai/docs/maas/capabilities/thinking (accessed 2026-06-25) family = "glm" release_date = "2026-01-06" last_updated = "2026-01-06" diff --git a/providers/google-vertex/models/zai-org/glm-5-maas.toml b/providers/google-vertex/models/zai-org/glm-5-maas.toml index f247fe0f8..bd0aa4f75 100644 --- a/providers/google-vertex/models/zai-org/glm-5-maas.toml +++ b/providers/google-vertex/models/zai-org/glm-5-maas.toml @@ -1,4 +1,7 @@ name = "GLM-5" +# POST .../endpoints/openapi/chat/completions; toggle with +# $.chat_template_kwargs.enable_thinking = true | false. +# https://cloud.google.com/vertex-ai/generative-ai/docs/maas/capabilities/thinking (accessed 2026-06-25) family = "glm" release_date = "2026-02-11" last_updated = "2026-02-11" diff --git a/providers/groq/models/openai/gpt-oss-120b.toml b/providers/groq/models/openai/gpt-oss-120b.toml index f2683f775..9d5cb0467 100644 --- a/providers/groq/models/openai/gpt-oss-120b.toml +++ b/providers/groq/models/openai/gpt-oss-120b.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.groq.com/openai/v1/chat/completions +# JSON reasoning_effort: "low" | "medium" | "high". +# Sources: https://console.groq.com/docs/reasoning#options-for-reasoning-effort-gptoss name = "GPT OSS 120B" family = "gpt-oss" release_date = "2025-08-05" diff --git a/providers/groq/models/openai/gpt-oss-20b.toml b/providers/groq/models/openai/gpt-oss-20b.toml index 1aa8095da..3dc79b4df 100644 --- a/providers/groq/models/openai/gpt-oss-20b.toml +++ b/providers/groq/models/openai/gpt-oss-20b.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.groq.com/openai/v1/chat/completions +# JSON reasoning_effort: "low" | "medium" | "high". +# Sources: https://console.groq.com/docs/reasoning#options-for-reasoning-effort-gptoss name = "GPT OSS 20B" family = "gpt-oss" release_date = "2025-08-05" diff --git a/providers/groq/models/openai/gpt-oss-safeguard-20b.toml b/providers/groq/models/openai/gpt-oss-safeguard-20b.toml index e96337e2d..75d8af019 100644 --- a/providers/groq/models/openai/gpt-oss-safeguard-20b.toml +++ b/providers/groq/models/openai/gpt-oss-safeguard-20b.toml @@ -1,3 +1,11 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.groq.com/openai/v1/chat/completions +# The model page says Harmony supports low/medium/high effort, but Groq's API +# reasoning page limits documented GPT-OSS reasoning_effort support to 20B/120B. +# Raw HTTP acceptance for this safeguard model remains unresolved. +# Sources: +# https://console.groq.com/docs/model/openai/gpt-oss-safeguard-20b +# https://console.groq.com/docs/reasoning#options-for-reasoning-effort-gptoss name = "Safety GPT OSS 20B" family = "gpt-oss" release_date = "2025-10-29" diff --git a/providers/groq/models/qwen/qwen3-32b.toml b/providers/groq/models/qwen/qwen3-32b.toml index 6b00faf6c..22897146a 100644 --- a/providers/groq/models/qwen/qwen3-32b.toml +++ b/providers/groq/models/qwen/qwen3-32b.toml @@ -1,3 +1,11 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.groq.com/openai/v1/chat/completions +# The model page recommends reasoning_effort: "default" for thinking and calls +# the model dual-mode, but the API page scopes "none"/"default" to Qwen 3.6 27B. +# Raw HTTP acceptance of both values for Qwen3-32B remains unresolved. +# Sources: +# https://console.groq.com/docs/model/qwen/qwen3-32b +# https://console.groq.com/docs/reasoning name = "Qwen3-32B" family = "qwen" release_date = "2025-06-11" diff --git a/providers/groq/provider.toml b/providers/groq/provider.toml index 51b8c3023..f9449d2c3 100644 --- a/providers/groq/provider.toml +++ b/providers/groq/provider.toml @@ -1,4 +1,11 @@ name = "Groq" env = ["GROQ_API_KEY"] npm = "@ai-sdk/groq" +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.groq.com/openai/v1/chat/completions +# JSON reasoning_effort is model-specific; reasoning_format: "parsed" | "raw" | +# "hidden" controls presentation, not reasoning. include_reasoning: true | false +# controls returned reasoning and is mutually exclusive with reasoning_format. +# Sources: +# https://console.groq.com/docs/reasoning doc = "https://console.groq.com/docs/models" \ No newline at end of file diff --git a/providers/helicone/models/claude-4.5-opus.toml b/providers/helicone/models/claude-4.5-opus.toml index 4954d3183..a9c4c365b 100644 --- a/providers/helicone/models/claude-4.5-opus.toml +++ b/providers/helicone/models/claude-4.5-opus.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Opus 4.5" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 63,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-opus" release_date = "2025-11-24" last_updated = "2025-11-24" diff --git a/providers/helicone/models/claude-4.5-sonnet.toml b/providers/helicone/models/claude-4.5-sonnet.toml index ea9e3bd68..50e91a4b4 100644 --- a/providers/helicone/models/claude-4.5-sonnet.toml +++ b/providers/helicone/models/claude-4.5-sonnet.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Sonnet 4.5" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 63,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-sonnet" release_date = "2025-09-29" last_updated = "2025-09-29" diff --git a/providers/helicone/models/claude-opus-4-1-20250805.toml b/providers/helicone/models/claude-opus-4-1-20250805.toml index 9debcb6e8..0aa88a1ec 100644 --- a/providers/helicone/models/claude-opus-4-1-20250805.toml +++ b/providers/helicone/models/claude-opus-4-1-20250805.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Opus 4.1 (20250805)" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 31,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-opus" release_date = "2025-08-05" last_updated = "2025-08-05" diff --git a/providers/helicone/models/claude-opus-4-1.toml b/providers/helicone/models/claude-opus-4-1.toml index bbad5e85f..16b8092bc 100644 --- a/providers/helicone/models/claude-opus-4-1.toml +++ b/providers/helicone/models/claude-opus-4-1.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Opus 4.1" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 31,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-opus" release_date = "2025-08-05" last_updated = "2025-08-05" diff --git a/providers/helicone/models/claude-opus-4.toml b/providers/helicone/models/claude-opus-4.toml index c11d49ba8..22ae60cfa 100644 --- a/providers/helicone/models/claude-opus-4.toml +++ b/providers/helicone/models/claude-opus-4.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Opus 4" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 31,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-opus" release_date = "2025-05-14" last_updated = "2025-05-14" diff --git a/providers/helicone/models/claude-sonnet-4-5-20250929.toml b/providers/helicone/models/claude-sonnet-4-5-20250929.toml index 1eed04f23..2ab06d84f 100644 --- a/providers/helicone/models/claude-sonnet-4-5-20250929.toml +++ b/providers/helicone/models/claude-sonnet-4-5-20250929.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Sonnet 4.5 (20250929)" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 63,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-sonnet" release_date = "2025-09-29" last_updated = "2025-09-29" diff --git a/providers/helicone/models/claude-sonnet-4.toml b/providers/helicone/models/claude-sonnet-4.toml index 45b85bd6b..1667703f8 100644 --- a/providers/helicone/models/claude-sonnet-4.toml +++ b/providers/helicone/models/claude-sonnet-4.toml @@ -1,4 +1,7 @@ name = "Anthropic: Claude Sonnet 4" +# Native exception (accessed 2026-06-25): `thinking.type` is enabled or disabled; +# manual `budget_tokens` is >=1024 and below max_tokens, hence this 63,999 cap. +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking family = "claude-sonnet" release_date = "2025-05-14" last_updated = "2025-05-14" diff --git a/providers/helicone/models/gemini-2.5-flash-lite.toml b/providers/helicone/models/gemini-2.5-flash-lite.toml index e8e32cea1..7d57cf400 100644 --- a/providers/helicone/models/gemini-2.5-flash-lite.toml +++ b/providers/helicone/models/gemini-2.5-flash-lite.toml @@ -1,4 +1,7 @@ name = "Google Gemini 2.5 Flash Lite" +# Native exception (accessed 2026-06-25): `thinkingBudget` 0 disables and -1 is +# dynamic; enabled manual budgets are 512..24576 tokens. +# https://ai.google.dev/gemini-api/docs/thinking family = "gemini-flash-lite" release_date = "2025-07-22" last_updated = "2025-07-22" diff --git a/providers/helicone/models/gemini-2.5-flash.toml b/providers/helicone/models/gemini-2.5-flash.toml index eeea25f14..3b459c57e 100644 --- a/providers/helicone/models/gemini-2.5-flash.toml +++ b/providers/helicone/models/gemini-2.5-flash.toml @@ -1,4 +1,7 @@ name = "Google Gemini 2.5 Flash" +# Native exception (accessed 2026-06-25): `thinkingBudget` 0 disables, -1 is +# dynamic, and the manual range is 0..24576 tokens. +# https://ai.google.dev/gemini-api/docs/thinking family = "gemini-flash" release_date = "2025-06-17" last_updated = "2025-06-17" diff --git a/providers/helicone/models/gemini-2.5-pro.toml b/providers/helicone/models/gemini-2.5-pro.toml index 355961f9b..e84304cbd 100644 --- a/providers/helicone/models/gemini-2.5-pro.toml +++ b/providers/helicone/models/gemini-2.5-pro.toml @@ -1,4 +1,7 @@ name = "Google Gemini 2.5 Pro" +# Native exception (accessed 2026-06-25): `thinkingBudget` -1 is dynamic and the +# manual range is 128..32768; thinking cannot be disabled with budget 0. +# https://ai.google.dev/gemini-api/docs/thinking family = "gemini-pro" release_date = "2025-06-17" last_updated = "2025-06-17" diff --git a/providers/helicone/models/gemini-3-pro-preview.toml b/providers/helicone/models/gemini-3-pro-preview.toml index 1a4fab47a..78e4b9915 100644 --- a/providers/helicone/models/gemini-3-pro-preview.toml +++ b/providers/helicone/models/gemini-3-pro-preview.toml @@ -1,4 +1,7 @@ name = "Google Gemini 3 Pro Preview" +# Native exception (accessed 2026-06-25): `thinkingLevel` supports low and high +# for Gemini 3 Pro; numeric `thinkingBudget` is a Gemini 2.5 control. +# https://ai.google.dev/gemini-api/docs/thinking family = "gemini-pro" release_date = "2025-11-18" last_updated = "2025-11-18" diff --git a/providers/helicone/models/gpt-oss-120b.toml b/providers/helicone/models/gpt-oss-120b.toml index fcc0d51d3..e8ee493ee 100644 --- a/providers/helicone/models/gpt-oss-120b.toml +++ b/providers/helicone/models/gpt-oss-120b.toml @@ -1,4 +1,7 @@ name = "OpenAI GPT-OSS 120b" +# Native exception (accessed 2026-06-25): top-level `reasoning_effort` is low, +# medium, or high; there is no per-request numeric reasoning-token budget. +# https://huggingface.co/openai/gpt-oss-120b family = "gpt-oss" release_date = "2024-06-01" last_updated = "2024-06-01" diff --git a/providers/helicone/models/gpt-oss-20b.toml b/providers/helicone/models/gpt-oss-20b.toml index ccf4d7dd8..f67ee4e73 100644 --- a/providers/helicone/models/gpt-oss-20b.toml +++ b/providers/helicone/models/gpt-oss-20b.toml @@ -1,4 +1,7 @@ name = "OpenAI GPT-OSS 20b" +# Native exception (accessed 2026-06-25): top-level `reasoning_effort` is low, +# medium, or high; there is no per-request numeric reasoning-token budget. +# https://huggingface.co/openai/gpt-oss-20b family = "gpt-oss" release_date = "2024-06-01" last_updated = "2024-06-01" diff --git a/providers/helicone/models/sonar-deep-research.toml b/providers/helicone/models/sonar-deep-research.toml index 069699a6c..36368fc53 100644 --- a/providers/helicone/models/sonar-deep-research.toml +++ b/providers/helicone/models/sonar-deep-research.toml @@ -1,4 +1,7 @@ name = "Perplexity Sonar Deep Research" +# Native exception (accessed 2026-06-25): `reasoning_effort` is minimal, low, +# medium, or high; no numeric reasoning-token budget is documented. +# https://docs.perplexity.ai/api-reference/sonar-post family = "sonar-deep-research" release_date = "2025-01-27" last_updated = "2025-01-27" diff --git a/providers/helicone/models/sonar-reasoning-pro.toml b/providers/helicone/models/sonar-reasoning-pro.toml index 5f12f1b08..28393f64b 100644 --- a/providers/helicone/models/sonar-reasoning-pro.toml +++ b/providers/helicone/models/sonar-reasoning-pro.toml @@ -1,4 +1,7 @@ name = "Perplexity Sonar Reasoning Pro" +# Native exception (accessed 2026-06-25): `reasoning_effort` is minimal, low, +# medium, or high; no numeric reasoning-token budget is documented. +# https://docs.perplexity.ai/api-reference/sonar-post family = "sonar-reasoning" release_date = "2025-01-27" last_updated = "2025-01-27" diff --git a/providers/helicone/provider.toml b/providers/helicone/provider.toml index 818fbfbfc..d781d5c21 100644 --- a/providers/helicone/provider.toml +++ b/providers/helicone/provider.toml @@ -1,5 +1,27 @@ name = "Helicone" env = ["HELICONE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions accepts top-level `reasoning_effort` = minimal, low, +# medium, or high and `reasoning_options.budget_tokens` = an integer from +# -9007199254740991 through 9007199254740991 (gateway schema, not model bounds). +# POST /v1/responses uses `reasoning.effort` plus the same nonstandard +# `reasoning_options.budget_tokens`; actual valid values and bounds are model-native. +# Anthropic routes translate any supplied effort to `thinking.type = "enabled"` +# and `thinking.budget_tokens` (explicit budget, else floor(max_tokens / 2)); +# Google routes translate to `generationConfig.thinkingConfig.thinkingLevel` for +# Gemini 3+, or `thinkingBudget = -1` for Gemini 2.5, then apply an explicit budget. +# These translators rebuild native payloads, stripping the two generic fields. +# Without effort, Gemini 2.5 gets `thinkingBudget = 0`; Gemini 3+ defaults to low. +# OpenAI routes pass native Chat/Responses shapes through. The Helicone-provider +# route sends GPT Pro/Codex to Responses even when the gateway request used Chat; +# other non-native Responses requests are converted through Chat first. +# Model comments record native exceptions, including disable sentinels and bounds. +# Sources: +# https://docs.helicone.ai/gateway/concepts/reasoning +# https://docs.helicone.ai/rest/ai-gateway/post-v1-chat-completions +# https://github.com/Helicone/helicone/blob/4df16a30ab79bc6f31e4b3a29aca179d767db878/packages/llm-mapper/transform/providers/openai/request/toAnthropic.ts +# https://github.com/Helicone/helicone/blob/4df16a30ab79bc6f31e4b3a29aca179d767db878/packages/llm-mapper/transform/providers/openai/request/toGoogle.ts +# https://github.com/Helicone/helicone/blob/4df16a30ab79bc6f31e4b3a29aca179d767db878/packages/cost/models/providers/helicone.ts api = "https://ai-gateway.helicone.ai/v1" doc = "https://helicone.ai/models" diff --git a/providers/hpc-ai/provider.toml b/providers/hpc-ai/provider.toml index 47d09e68a..7910346fc 100644 --- a/providers/hpc-ai/provider.toml +++ b/providers/hpc-ai/provider.toml @@ -1,5 +1,9 @@ name = "HPC-AI" env = ["HPC_AI_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): POST /inference/v1/chat/completions +# accepts top-level `reasoning_effort` = none, low, medium, or high. The reference +# documents no numeric reasoning budget; support among models is not specified. +# https://www.hpc-ai.com/doc/docs/Model-APIs/API-Reference/OpenAI-Compatible-API/Create-Chat-Completions/ doc = "https://www.hpc-ai.com/doc/docs/quickstart/" api = "https://api.hpc-ai.com/inference/v1" diff --git a/providers/huggingface/provider.toml b/providers/huggingface/provider.toml index 234d8fdf7..5e902ce04 100644 --- a/providers/huggingface/provider.toml +++ b/providers/huggingface/provider.toml @@ -1,5 +1,11 @@ name = "Hugging Face" env = ["HF_TOKEN"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/chat/completions accepts +# top-level `reasoning_effort`; common values are none, minimal, low, medium, +# high, and xhigh. Support, defaults, and meaningful values depend on both the +# selected model and the inference provider chosen by HF routing. No shared +# toggle field or numeric reasoning-budget field is documented. +# https://huggingface.co/docs/inference-providers/en/tasks/chat-completion api = "https://router.huggingface.co/v1" doc = "https://huggingface.co/docs/inference-providers" diff --git a/providers/iflowcn/provider.toml b/providers/iflowcn/provider.toml index 610dbe44f..56482a162 100644 --- a/providers/iflowcn/provider.toml +++ b/providers/iflowcn/provider.toml @@ -1,5 +1,12 @@ name = "iFlow" env = ["IFLOW_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): the retired iFlow CLI advertised +# a UI-level "Thinking" feature and configured `https://apis.iflow.cn/v1`, but +# that is not documentation of a raw POST /v1/chat/completions request field. +# Current platform API docs cover Search endpoints only and publish no raw LLM +# toggle, effort, or reasoning-token budget field for this configured API. +# https://github.com/iflow-ai/iflow-cli/blob/main/README.md +# https://platform.iflow.cn/en/docs/api-reference doc = "https://platform.iflow.cn/en/docs" api = "https://apis.iflow.cn/v1" diff --git a/providers/inception/models/mercury-2.toml b/providers/inception/models/mercury-2.toml index fdba0c88e..672d91c9b 100644 --- a/providers/inception/models/mercury-2.toml +++ b/providers/inception/models/mercury-2.toml @@ -1,4 +1,7 @@ name = "Mercury 2" +# Model exception (accessed 2026-06-25): `reasoning_effort = "instant"` is also +# accepted for near-realtime responses, but it is not a models.dev effort enum. +# https://docs.inceptionlabs.ai/capabilities/instant family = "mercury" release_date = "2026-02-24" last_updated = "2026-02-24" diff --git a/providers/inception/models/mercury-edit-2.toml b/providers/inception/models/mercury-edit-2.toml index 7395aab8a..768b64300 100644 --- a/providers/inception/models/mercury-edit-2.toml +++ b/providers/inception/models/mercury-edit-2.toml @@ -1,4 +1,7 @@ name = "Mercury Edit 2" +# Model exception (accessed 2026-06-25): the /v1/edit/completions and +# /v1/fim/completions request schemas expose no reasoning-control field. +# https://api.inceptionlabs.ai/openapi.json release_date = "2026-03-30" last_updated = "2026-03-30" attachment = false diff --git a/providers/inception/provider.toml b/providers/inception/provider.toml index cdd686a9c..362d02d8f 100644 --- a/providers/inception/provider.toml +++ b/providers/inception/provider.toml @@ -1,5 +1,11 @@ name = "Inception" env = ["INCEPTION_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/chat/completions accepts +# top-level `reasoning_effort` = instant, low, medium, or high; default medium. +# It documents no off value or numeric reasoning-token budget. `instant` is the +# near-realtime Mercury 2 mode for low-latency turns that do not require tools. +# https://docs.inceptionlabs.ai/api-reference/chat/create-a-chat-completion +# https://docs.inceptionlabs.ai/capabilities/instant api = "https://api.inceptionlabs.ai/v1/" doc = "https://platform.inceptionlabs.ai/docs" diff --git a/providers/inceptron/provider.toml b/providers/inceptron/provider.toml index f6f96d7f0..bea239cbb 100644 --- a/providers/inceptron/provider.toml +++ b/providers/inceptron/provider.toml @@ -1,5 +1,9 @@ name = "Inceptron" npm = "@ai-sdk/openai-compatible" env = ["INCEPTRON_API_KEY"] +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/chat/completions. The +# provider's complete request table documents no toggle, effort, or numeric +# reasoning-token budget field. +# https://docs.inceptron.io/API%20Reference/chat-completions api = "https://api.inceptron.io/v1" doc = "https://docs.inceptron.io" diff --git a/providers/inference/provider.toml b/providers/inference/provider.toml index 40be66ddd..0a99753d6 100644 --- a/providers/inference/provider.toml +++ b/providers/inference/provider.toml @@ -1,5 +1,9 @@ name = "Inference" env = ["INFERENCE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): POST /v1/chat/completions. The +# provider's supported-parameter table documents no toggle, effort, or numeric +# reasoning-token budget field; it directs users to request unlisted fields. +# https://docs.inference.net/api/api-quickstart api = "https://inference.net/v1" doc = "https://inference.net/models" diff --git a/providers/io-net/provider.toml b/providers/io-net/provider.toml index e758c4c00..a8f0f2333 100644 --- a/providers/io-net/provider.toml +++ b/providers/io-net/provider.toml @@ -1,5 +1,11 @@ name = "IO.NET" env = ["IOINTELLIGENCE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): POST /api/v1/chat/completions uses +# `reasoning.effort` = none, low, medium, or high; none disables and omission uses +# the model default. Legacy top-level `reasoning_effort` accepts low, medium, or +# high. Always-reasoning models may reject or ignore none. No token budget field +# is documented; response reasoning is `choices[0].message.reasoning_content`. +# https://io.net/docs/guides/intelligence/reasoning-for-chat-completions doc = "https://io.net/docs/guides/intelligence/io-intelligence" api = "https://api.intelligence.io.solutions/api/v1" diff --git a/providers/jiekou/provider.toml b/providers/jiekou/provider.toml index 37d33fad1..b3b922df6 100644 --- a/providers/jiekou/provider.toml +++ b/providers/jiekou/provider.toml @@ -1,5 +1,21 @@ name = "Jiekou.AI" env = ["JIEKOU_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP formats (accessed 2026-06-25; docs use api.highwayapi.ai): +# OpenAI: POST /openai/chat/completions uses top-level `reasoning_effort`. +# For Gemini it maps disable|none -> budget 0, low -> 1024, medium -> 2048, +# high -> 4096; Gemini 2.5 Pro maps none to its minimum 128 instead of off. +# Gemini: POST /gemini/v1/models/{model}:generateContent uses +# `generationConfig.thinkingConfig.thinkingBudget`; stream uses +# `{model}:streamGenerateContent`. Native sentinels are 0 = off and -1 = dynamic; +# 2.5 Pro is 128..32768, Flash is 0..24576, and Flash-Lite is 512..24576 plus 0. +# Anthropic: POST /anthropic/v1/messages uses `thinking.type = "enabled"` and +# integer `thinking.budget_tokens`; Jiekou documents control only on this native +# route. The native minimum is 1024 and manual budgets must be below `max_tokens`. +# Sources: +# https://docs.jiekou.ai/docs/providers/gemini +# https://docs.jiekou.ai/docs/providers/anthropic +# https://ai.google.dev/gemini-api/docs/thinking +# https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking api = "https://api.jiekou.ai/openai" doc = "https://docs.jiekou.ai/docs/support/quickstart?utm_source=github_models.dev" diff --git a/providers/kilo/provider.toml b/providers/kilo/provider.toml index 6ff612335..5421ba25c 100644 --- a/providers/kilo/provider.toml +++ b/providers/kilo/provider.toml @@ -1,5 +1,24 @@ name = "Kilo Gateway" env = ["KILO_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP formats (accessed 2026-06-25): Kilo accepts POST +# /api/gateway/chat/completions, /responses, and /messages (also under /v1). +# Chat uses `reasoning.enabled` = true|false, `reasoning.effort` = none, minimal, +# low, medium, high, xhigh, or max, and integer `reasoning.max_tokens`; legacy +# top-level `reasoning_effort` is also read. none is the disable sentinel. +# Responses uses `reasoning.effort` with the same model-dependent effort set; +# none disables and any other supplied value enables. +# Messages uses `thinking.type` = enabled|adaptive|disabled and, for enabled +# thinking, integer `thinking.budget_tokens` >=1024 and below `max_tokens`; +# Anthropic `output_config.effort` = low, medium, high, or max. Exact accepted +# subsets and maximum budgets remain model-specific. These are route-native +# mappings, not one JSON shape forwarded unchanged to every route. +# Sources: +# https://kilo.ai/docs/gateway +# https://github.com/Kilo-Org/cloud/blob/deec94c0a6515cfeaf5748993ad9b2d601921e78/apps/web/src/app/api/openrouter/%5B...path%5D/route.ts +# https://github.com/Kilo-Org/cloud/blob/deec94c0a6515cfeaf5748993ad9b2d601921e78/apps/web/src/lib/ai-gateway/providers/openrouter/request-helpers.ts +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +# https://platform.openai.com/docs/api-reference/responses/create#responses-create-reasoning +# https://docs.anthropic.com/en/api/messages api = "https://api.kilo.ai/api/gateway" doc = "https://kilo.ai" diff --git a/providers/kuae-cloud-coding-plan/models/GLM-4.7.toml b/providers/kuae-cloud-coding-plan/models/GLM-4.7.toml index d51f8d99e..19756b729 100644 --- a/providers/kuae-cloud-coding-plan/models/GLM-4.7.toml +++ b/providers/kuae-cloud-coding-plan/models/GLM-4.7.toml @@ -1,3 +1,6 @@ +# POST /v1/chat/completions shows $.thinking.type = "enabled" only; it does not +# document a disable value for the same model ID, so a full toggle is unproven. +# https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/ (accessed 2026-06-25) name = "GLM-4.7" family = "glm" release_date = "2025-12-22" diff --git a/providers/kuae-cloud-coding-plan/provider.toml b/providers/kuae-cloud-coding-plan/provider.toml index 2776c1143..9a3e46dc6 100644 --- a/providers/kuae-cloud-coding-plan/provider.toml +++ b/providers/kuae-cloud-coding-plan/provider.toml @@ -1,3 +1,6 @@ +# Raw HTTP uses POST /v1/chat/completions; the Coding Plan guide does not +# publish effort values or a reasoning-token budget in its request surface. +# https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/ (accessed 2026-06-25) name = "KUAE Cloud Coding Plan" env = ["KUAE_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/lilac/models/google/gemma-4-31b-it.toml b/providers/lilac/models/google/gemma-4-31b-it.toml index 50732a3a4..2e1b61ddc 100644 --- a/providers/lilac/models/google/gemma-4-31b-it.toml +++ b/providers/lilac/models/google/gemma-4-31b-it.toml @@ -1,3 +1,6 @@ +# $.chat_template_kwargs.enable_thinking = true|false; default false. The +# template ignores $.chat_template_kwargs.thinking when sent alone. +# https://docs.getlilac.com/inference/chat-completions#reasoning (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }] base_model = "google/gemma-4-31b-it" knowledge = "2025-01" diff --git a/providers/lilac/models/minimaxai/minimax-m3.toml b/providers/lilac/models/minimaxai/minimax-m3.toml index b02a987a9..93b80d0ae 100644 --- a/providers/lilac/models/minimaxai/minimax-m3.toml +++ b/providers/lilac/models/minimaxai/minimax-m3.toml @@ -1,3 +1,6 @@ +# $.chat_template_kwargs.thinking_mode = "adaptive"|"enabled"|"disabled"; +# adaptive is the default, enabled always thinks, and disabled never thinks. +# https://docs.getlilac.com/inference/chat-completions#reasoning (accessed 2026-06-25) base_model = "minimax/MiniMax-M3" name = "MiniMax M3" family = "minimax-m3" diff --git a/providers/lilac/models/moonshotai/kimi-k2.6.toml b/providers/lilac/models/moonshotai/kimi-k2.6.toml index a6457870b..9d1b12cd8 100644 --- a/providers/lilac/models/moonshotai/kimi-k2.6.toml +++ b/providers/lilac/models/moonshotai/kimi-k2.6.toml @@ -1,3 +1,6 @@ +# $.chat_template_kwargs.thinking = true|false; default true. The similarly +# named enable_thinking key is ignored by this Moonshot template. +# https://docs.getlilac.com/inference/chat-completions#reasoning (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }] base_model = "moonshotai/kimi-k2.6" diff --git a/providers/lilac/models/zai-org/glm-5.2.toml b/providers/lilac/models/zai-org/glm-5.2.toml index 51c6c19c6..c2f815106 100644 --- a/providers/lilac/models/zai-org/glm-5.2.toml +++ b/providers/lilac/models/zai-org/glm-5.2.toml @@ -1,3 +1,7 @@ +# Toggle: $.chat_template_kwargs.enable_thinking = true|false (default true). +# Effort: $.reasoning_effort or $.chat_template_kwargs.reasoning_effort is +# "max" (default) or "high"; effort has no effect when thinking is disabled. +# https://docs.getlilac.com/inference/chat-completions#reasoning (accessed 2026-06-25) base_model = "zhipuai/glm-5.2" name = "GLM 5.2" reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] diff --git a/providers/lilac/provider.toml b/providers/lilac/provider.toml index d19330715..319827342 100644 --- a/providers/lilac/provider.toml +++ b/providers/lilac/provider.toml @@ -1,3 +1,6 @@ +# POST /v1/chat/completions carries model-template controls under +# $.chat_template_kwargs; keys and values are model-specific, not universal. +# https://docs.getlilac.com/inference/chat-completions (accessed 2026-06-25) name = "Lilac" env = ["LILAC_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/llama/provider.toml b/providers/llama/provider.toml index b8b7e6705..1faed2ee0 100644 --- a/providers/llama/provider.toml +++ b/providers/llama/provider.toml @@ -1,3 +1,6 @@ +# POST /compat/v1/chat/completions follows Meta's published Chat request +# schema, which lists no toggle, effort, or reasoning-token-budget field. +# https://llama.developer.meta.com/docs/api-reference/chat-completions (accessed 2026-06-25) name = "Llama" env = ["LLAMA_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/llmgateway/models/grok-4-1-fast-reasoning.toml b/providers/llmgateway/models/grok-4-1-fast-reasoning.toml index 4c9094c98..b2cf5e5cc 100644 --- a/providers/llmgateway/models/grok-4-1-fast-reasoning.toml +++ b/providers/llmgateway/models/grok-4-1-fast-reasoning.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) name = "Grok 4.1 Fast Reasoning" family = "grok" release_date = "2025-11-19" diff --git a/providers/llmgateway/models/grok-4-20-beta-0309-reasoning.toml b/providers/llmgateway/models/grok-4-20-beta-0309-reasoning.toml index 72bd3881f..14341f1d0 100644 --- a/providers/llmgateway/models/grok-4-20-beta-0309-reasoning.toml +++ b/providers/llmgateway/models/grok-4-20-beta-0309-reasoning.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xai/grok-4.20-0309-reasoning" [[reasoning_options]] diff --git a/providers/llmgateway/models/grok-4-20-reasoning.toml b/providers/llmgateway/models/grok-4-20-reasoning.toml index 72bd3881f..14341f1d0 100644 --- a/providers/llmgateway/models/grok-4-20-reasoning.toml +++ b/providers/llmgateway/models/grok-4-20-reasoning.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xai/grok-4.20-0309-reasoning" [[reasoning_options]] diff --git a/providers/llmgateway/models/grok-4-3.toml b/providers/llmgateway/models/grok-4-3.toml index f4890c2cd..45588ea58 100644 --- a/providers/llmgateway/models/grok-4-3.toml +++ b/providers/llmgateway/models/grok-4-3.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xai/grok-4.3" [[reasoning_options]] diff --git a/providers/llmgateway/models/grok-build-0-1.toml b/providers/llmgateway/models/grok-build-0-1.toml index 3b75a2441..22fe42564 100644 --- a/providers/llmgateway/models/grok-build-0-1.toml +++ b/providers/llmgateway/models/grok-build-0-1.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xai/grok-build-0.1" reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/llmgateway/models/kimi-k2.7-code-highspeed.toml b/providers/llmgateway/models/kimi-k2.7-code-highspeed.toml index b153a3ed7..43e21e5dc 100644 --- a/providers/llmgateway/models/kimi-k2.7-code-highspeed.toml +++ b/providers/llmgateway/models/kimi-k2.7-code-highspeed.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "moonshotai/kimi-k2.7-code-highspeed" reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/llmgateway/models/kimi-k2.7-code.toml b/providers/llmgateway/models/kimi-k2.7-code.toml index 1967bf56e..df4e4bb07 100644 --- a/providers/llmgateway/models/kimi-k2.7-code.toml +++ b/providers/llmgateway/models/kimi-k2.7-code.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "moonshotai/kimi-k2.7-code" reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/llmgateway/models/mimo-v2-pro.toml b/providers/llmgateway/models/mimo-v2-pro.toml index fbb9a1df6..22c1ed6ab 100644 --- a/providers/llmgateway/models/mimo-v2-pro.toml +++ b/providers/llmgateway/models/mimo-v2-pro.toml @@ -1,3 +1,6 @@ +# Gateway-wide $.reasoning_effort = "none" disables non-OpenAI reasoning; +# this model has no documented effort tiers or exact reasoning-token budget. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xiaomi/mimo-v2-pro" [[reasoning_options]] diff --git a/providers/llmgateway/models/mimo-v2.5-pro.toml b/providers/llmgateway/models/mimo-v2.5-pro.toml index 74b39fe7c..92a81b100 100644 --- a/providers/llmgateway/models/mimo-v2.5-pro.toml +++ b/providers/llmgateway/models/mimo-v2.5-pro.toml @@ -1,3 +1,6 @@ +# Gateway-wide $.reasoning_effort = "none" disables non-OpenAI reasoning; +# this model has no documented effort tiers or exact reasoning-token budget. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xiaomi/mimo-v2.5-pro" [[reasoning_options]] diff --git a/providers/llmgateway/models/mimo-v2.5.toml b/providers/llmgateway/models/mimo-v2.5.toml index 7409f3efa..69b3d76b2 100644 --- a/providers/llmgateway/models/mimo-v2.5.toml +++ b/providers/llmgateway/models/mimo-v2.5.toml @@ -1,3 +1,6 @@ +# Gateway-wide $.reasoning_effort = "none" disables non-OpenAI reasoning; +# this model has no documented effort tiers or exact reasoning-token budget. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "xiaomi/mimo-v2.5" [[reasoning_options]] diff --git a/providers/llmgateway/models/nemotron-3-ultra-550b.toml b/providers/llmgateway/models/nemotron-3-ultra-550b.toml index 0c13e8388..e89725401 100644 --- a/providers/llmgateway/models/nemotron-3-ultra-550b.toml +++ b/providers/llmgateway/models/nemotron-3-ultra-550b.toml @@ -1,3 +1,6 @@ +# Gateway-wide $.reasoning_effort = "none" disables non-OpenAI reasoning; +# this model has no documented effort tiers or exact reasoning-token budget. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "nvidia/nemotron-3-ultra-550b-a55b" [[reasoning_options]] diff --git a/providers/llmgateway/models/sonar-reasoning-pro.toml b/providers/llmgateway/models/sonar-reasoning-pro.toml index b19b3ee4b..1227bc6c1 100644 --- a/providers/llmgateway/models/sonar-reasoning-pro.toml +++ b/providers/llmgateway/models/sonar-reasoning-pro.toml @@ -1,3 +1,6 @@ +# This model is restricted here to $.reasoning_effort or $.reasoning.effort +# = "low"|"medium"|"high"; no exact token budget is documented for it. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) base_model = "perplexity/sonar-reasoning-pro" reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] diff --git a/providers/llmgateway/provider.toml b/providers/llmgateway/provider.toml index 7f3a03eac..5381d21a8 100644 --- a/providers/llmgateway/provider.toml +++ b/providers/llmgateway/provider.toml @@ -1,3 +1,11 @@ +# POST /v1/chat/completions accepts $.reasoning_effort = none|minimal|low| +# medium|high|xhigh|max. Its raw schema lists $.reasoning.effort = low|medium| +# high; the two effort paths are mutually exclusive. $.reasoning.max_tokens +# overrides either effort path. Anthropic budgets are clamped to 1024..128000. +# POST /v1/messages translates $.thinking to unified reasoning controls; its +# $.output_config.effort controls adaptive depth on Opus 4.7+. +# https://docs.llmgateway.io/features/reasoning (accessed 2026-06-25) +# https://docs.llmgateway.io/v1_messages (accessed 2026-06-25) name = "LLM Gateway" env = ["LLMGATEWAY_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/llmtr/models/qwen3-6-35b.toml b/providers/llmtr/models/qwen3-6-35b.toml index 5fad3173d..f9d4ddfad 100644 --- a/providers/llmtr/models/qwen3-6-35b.toml +++ b/providers/llmtr/models/qwen3-6-35b.toml @@ -1,3 +1,6 @@ +# LLMTR's endpoint-specific reasoning pages document Responses effort and GLM +# toggles, but no raw Qwen3.6-35B toggle field or values; this claim is unproven. +# https://llmtr.com/docs/gateway/reasoning-effort/ (accessed 2026-06-25) base_model = "alibaba/qwen3.6-35b-a3b" reasoning_options = [{ type = "toggle" }] diff --git a/providers/llmtr/provider.toml b/providers/llmtr/provider.toml index 5e769c633..eeaa37b3c 100644 --- a/providers/llmtr/provider.toml +++ b/providers/llmtr/provider.toml @@ -1,3 +1,12 @@ +# POST /v1/responses uses $.reasoning.effort = minimal|low|medium|high|xhigh; +# model suffixes :min|:low|:medium|:med|:high|:max|:xhigh are aliases, body +# wins, and unsupported levels return 400 unsupported_capability. Codex models +# are Responses-only and return 400 endpoint_mismatch at /v1/chat/completions. +# Chat GLM thinking instead uses $.reasoning = true|false or :think|:fast; +# omission sends upstream $.thinking.type = "disabled". No token-budget field +# is published for either endpoint. +# https://llmtr.com/docs/gateway/reasoning-effort/ (accessed 2026-06-25) +# https://llmtr.com/docs/gateway/zai-thinking/ (accessed 2026-06-25) name = "LLMTR" env = ["LLMTR_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/lmstudio/models/openai/gpt-oss-20b.toml b/providers/lmstudio/models/openai/gpt-oss-20b.toml index cf64b5998..e883c2043 100644 --- a/providers/lmstudio/models/openai/gpt-oss-20b.toml +++ b/providers/lmstudio/models/openai/gpt-oss-20b.toml @@ -1,3 +1,6 @@ +# low|medium|high is documented at $.reasoning.effort on /v1/responses, not on +# this provider's configured /v1 Chat Completions surface; no budget is listed. +# https://lmstudio.ai/docs/developer/openai-compat/responses (accessed 2026-06-25) name = "GPT OSS 20B" family = "gpt-oss" release_date = "2025-08-05" diff --git a/providers/lmstudio/provider.toml b/providers/lmstudio/provider.toml index 038912bf1..686f1f7e6 100644 --- a/providers/lmstudio/provider.toml +++ b/providers/lmstudio/provider.toml @@ -1,3 +1,9 @@ +# Native POST /api/v1/chat: $.reasoning = "off"|"low"|"medium"|"high"|"on". +# OpenAI POST /v1/responses: $.reasoning.effort (the example uses "low"). +# The published POST /v1/chat/completions payload list has no reasoning field. +# https://lmstudio.ai/docs/developer/rest/chat (accessed 2026-06-25) +# https://lmstudio.ai/docs/developer/openai-compat/responses (accessed 2026-06-25) +# https://lmstudio.ai/docs/developer/openai-compat/chat-completions (accessed 2026-06-25) name = "LMStudio" env = ["LMSTUDIO_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/lucidquery/provider.toml b/providers/lucidquery/provider.toml index 17d0c2b18..03746e9de 100644 --- a/providers/lucidquery/provider.toml +++ b/providers/lucidquery/provider.toml @@ -1,3 +1,7 @@ +# POST /v1/chat/completions enables a trace with $.thinking = true; aliases are +# $.include_reasoning and $.reasoning.enabled = true. The published request +# schema gives no false value, effort enum, or reasoning-token budget. +# https://lucidquery.com/docs#reasoning (accessed 2026-06-25) name = "LucidQuery" env = ["LUCIDQUERY_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/meganova/models/zai-org/GLM-5.toml b/providers/meganova/models/zai-org/GLM-5.toml index 6704497bf..25c3ed04e 100644 --- a/providers/meganova/models/zai-org/GLM-5.toml +++ b/providers/meganova/models/zai-org/GLM-5.toml @@ -1,3 +1,6 @@ +# MegaNova identifies GLM-5 as a reasoning model, but its raw Chat request +# schema documents no field or values that enable and disable reasoning. +# https://docs.meganova.ai/inference-models/text-generation.md (accessed 2026-06-25) name = "GLM-5" family = "glm" release_date = "2026-02-11" diff --git a/providers/meganova/provider.toml b/providers/meganova/provider.toml index 1afd1ae49..0bdcc7190 100644 --- a/providers/meganova/provider.toml +++ b/providers/meganova/provider.toml @@ -1,3 +1,6 @@ +# POST /v1/chat/completions publishes messages, model, max_tokens, temperature, +# top_p, and stream; no toggle, effort, or reasoning-token budget is listed. +# https://docs.meganova.ai/inference-models/text-generation.md (accessed 2026-06-25) name = "Meganova" env = ["MEGANOVA_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/merge-gateway/provider.toml b/providers/merge-gateway/provider.toml index e9c037abe..93c78884b 100644 --- a/providers/merge-gateway/provider.toml +++ b/providers/merge-gateway/provider.toml @@ -1,4 +1,20 @@ name = "Merge Gateway" env = ["MERGE_GATEWAY_API_KEY"] npm = "merge-gateway-ai-sdk-provider" +# Reasoning request surfaces (sources accessed 2026-06-25): +# - OpenAI compatibility: POST /v1/openai/chat/completions uses the native +# top-level `reasoning_effort`; the documented base accepts OpenAI SDK calls. +# - Anthropic compatibility: POST /v1/anthropic/v1/messages uses native +# `thinking = { type = "enabled"|"disabled", budget_tokens = N }` and +# `output_config.effort`. +# https://docs.merge.dev/merge-gateway/get-started +# https://docs.anthropic.com/en/api/messages +# - This npm package calls POST /v1/ai-sdk/chat/completions and maps +# `providerOptions.mergeGateway.thinking` to +# `thinking = { type = "enabled"|"disabled", budget_tokens = N }`. +# https://github.com/merge-api/merge-gateway-ai-sdk-provider/blob/main/src/chat/index.ts +# - Native POST /v1/responses has no reasoning field in its strict request +# schema. Other normalized or dynamically routed routes do not document +# transparent native-field passthrough; do not infer support from them. +# https://docs.merge.dev/merge-gateway/api-overview/responses/create doc = "https://docs.merge.dev/merge-gateway" diff --git a/providers/mixlayer/models/qwen/qwen3.5-122b-a10b.toml b/providers/mixlayer/models/qwen/qwen3.5-122b-a10b.toml index bf923a15e..5ab6f2bca 100644 --- a/providers/mixlayer/models/qwen/qwen3.5-122b-a10b.toml +++ b/providers/mixlayer/models/qwen/qwen3.5-122b-a10b.toml @@ -1,3 +1,5 @@ +# Qwen 3.5 supports the provider's binary $.thinking = true|false mapping. +# https://docs.mixlayer.com/reasoning (accessed 2026-06-25) name = "Qwen3.5 122B A10B" family = "qwen" release_date = "2026-03-18" diff --git a/providers/mixlayer/models/qwen/qwen3.5-27b.toml b/providers/mixlayer/models/qwen/qwen3.5-27b.toml index 7667c5c97..7ced61596 100644 --- a/providers/mixlayer/models/qwen/qwen3.5-27b.toml +++ b/providers/mixlayer/models/qwen/qwen3.5-27b.toml @@ -1,3 +1,5 @@ +# Qwen 3.5 supports the provider's binary $.thinking = true|false mapping. +# https://docs.mixlayer.com/reasoning (accessed 2026-06-25) name = "Qwen3.5 27B" family = "qwen" release_date = "2026-03-18" diff --git a/providers/mixlayer/models/qwen/qwen3.5-35b-a3b.toml b/providers/mixlayer/models/qwen/qwen3.5-35b-a3b.toml index 73c04a5b0..047af66f5 100644 --- a/providers/mixlayer/models/qwen/qwen3.5-35b-a3b.toml +++ b/providers/mixlayer/models/qwen/qwen3.5-35b-a3b.toml @@ -1,3 +1,5 @@ +# Qwen 3.5 supports the provider's binary $.thinking = true|false mapping. +# https://docs.mixlayer.com/reasoning (accessed 2026-06-25) name = "Qwen3.5 35B A3B" family = "qwen" release_date = "2026-03-18" diff --git a/providers/mixlayer/models/qwen/qwen3.5-397b-a17b.toml b/providers/mixlayer/models/qwen/qwen3.5-397b-a17b.toml index 60421c7e5..54af356e6 100644 --- a/providers/mixlayer/models/qwen/qwen3.5-397b-a17b.toml +++ b/providers/mixlayer/models/qwen/qwen3.5-397b-a17b.toml @@ -1,3 +1,5 @@ +# Qwen 3.5 supports the provider's binary $.thinking = true|false mapping. +# https://docs.mixlayer.com/reasoning (accessed 2026-06-25) name = "Qwen3.5 397B A17B" family = "qwen" release_date = "2026-03-18" diff --git a/providers/mixlayer/models/qwen/qwen3.5-9b.toml b/providers/mixlayer/models/qwen/qwen3.5-9b.toml index 07b9381cc..2213cbf29 100644 --- a/providers/mixlayer/models/qwen/qwen3.5-9b.toml +++ b/providers/mixlayer/models/qwen/qwen3.5-9b.toml @@ -1,3 +1,5 @@ +# Qwen 3.5 supports the provider's binary $.thinking = true|false mapping. +# https://docs.mixlayer.com/reasoning (accessed 2026-06-25) name = "Qwen3.5 9B" family = "qwen" release_date = "2026-03-18" diff --git a/providers/mixlayer/provider.toml b/providers/mixlayer/provider.toml index d283d3741..591d6cf4a 100644 --- a/providers/mixlayer/provider.toml +++ b/providers/mixlayer/provider.toml @@ -1,3 +1,7 @@ +# POST /v1/chat/completions uses $.thinking = true|false. The accepted alias +# $.reasoning_effort = "low"|"medium"|"high" maps only to binary enablement; +# the level is currently ignored/reserved, so it is not an effort control. +# https://docs.mixlayer.com/reasoning (accessed 2026-06-25) name = "Mixlayer" env = ["MIXLAYER_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/moark/models/GLM-4.7.toml b/providers/moark/models/GLM-4.7.toml index a33fa9453..81ebd666d 100644 --- a/providers/moark/models/GLM-4.7.toml +++ b/providers/moark/models/GLM-4.7.toml @@ -1,3 +1,6 @@ +# The Moark raw Chat schema exposes no user-selectable reasoning control for +# this reasoning-only model. +# https://moark.com/docs/openapi/v1 (accessed 2026-06-25) name = "GLM-4.7" family = "glm" release_date = "2025-12-22" diff --git a/providers/moark/models/MiniMax-M2.1.toml b/providers/moark/models/MiniMax-M2.1.toml index 1f98c7a16..1d4584699 100644 --- a/providers/moark/models/MiniMax-M2.1.toml +++ b/providers/moark/models/MiniMax-M2.1.toml @@ -1,3 +1,6 @@ +# The Moark raw Chat schema exposes no user-selectable reasoning control for +# this reasoning-only model. +# https://moark.com/docs/openapi/v1 (accessed 2026-06-25) name = "MiniMax-M2.1" family = "minimax" release_date = "2025-12-23" diff --git a/providers/moark/provider.toml b/providers/moark/provider.toml index a714dc468..bb66b171b 100644 --- a/providers/moark/provider.toml +++ b/providers/moark/provider.toml @@ -1,3 +1,6 @@ +# POST /v1/chat/completions is exposed by Moark's raw OpenAPI reference; its +# request schema documents no toggle, effort, or reasoning-token-budget field. +# https://moark.com/docs/openapi/v1 (accessed 2026-06-25) name = "Moark" npm = "@ai-sdk/openai-compatible" env = ["MOARK_API_KEY"] diff --git a/providers/modelscope/provider.toml b/providers/modelscope/provider.toml index 703583374..c55d6f76c 100644 --- a/providers/modelscope/provider.toml +++ b/providers/modelscope/provider.toml @@ -1,5 +1,8 @@ name = "ModelScope" env = ["MODELSCOPE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP is POST `/v1/chat/completions`. ModelScope's API Inference docs do +# not document a shared reasoning toggle, effort, or token-budget field. +# https://modelscope.cn/docs/model-service/API-Inference/intro (accessed 2026-06-25) api = "https://api-inference.modelscope.cn/v1" doc = "https://modelscope.cn/docs/model-service/API-Inference/intro" \ No newline at end of file diff --git a/providers/moonshotai-cn/provider.toml b/providers/moonshotai-cn/provider.toml index 3aa6f1fe1..e92e23572 100644 --- a/providers/moonshotai-cn/provider.toml +++ b/providers/moonshotai-cn/provider.toml @@ -1,5 +1,9 @@ name = "Moonshot AI (China)" env = ["MOONSHOT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Chat is POST `/v1/chat/completions`. Kimi K2.5 and K2.6 toggle reasoning with +# `thinking.type = "enabled"|"disabled"` (enabled by default); K2.7 Code only +# accepts `"enabled"`, so it is always on. No effort or token budget is exposed. +# https://platform.kimi.com/docs/api/chat (accessed 2026-06-25) doc = "https://platform.moonshot.cn/docs/api/chat" api = "https://api.moonshot.cn/v1" diff --git a/providers/moonshotai/provider.toml b/providers/moonshotai/provider.toml index bc1120357..214d04ea5 100644 --- a/providers/moonshotai/provider.toml +++ b/providers/moonshotai/provider.toml @@ -1,5 +1,9 @@ name = "Moonshot AI" env = ["MOONSHOT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Chat is POST `/v1/chat/completions`. Kimi K2.5 and K2.6 toggle reasoning with +# `thinking.type = "enabled"|"disabled"` (enabled by default); K2.7 Code only +# accepts `"enabled"`, so it is always on. No effort or token budget is exposed. +# https://platform.kimi.ai/docs/api/chat (accessed 2026-06-25) doc = "https://platform.moonshot.ai/docs/api/chat" api = "https://api.moonshot.ai/v1" diff --git a/providers/morph/provider.toml b/providers/morph/provider.toml index 512cc2b45..7febf5783 100644 --- a/providers/morph/provider.toml +++ b/providers/morph/provider.toml @@ -1,5 +1,9 @@ name = "Morph" env = ["MORPH_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP for the configured Apply models is POST `/v1/chat/completions`. +# Morph documents no reasoning toggle, effort, or token-budget request field +# for `morph-v3-fast`, `morph-v3-large`, or `auto`. +# https://docs.morphllm.com/llms.txt (accessed 2026-06-25) api = "https://api.morphllm.com/v1" doc = "https://docs.morphllm.com/api-reference/introduction" \ No newline at end of file diff --git a/providers/nano-gpt/provider.toml b/providers/nano-gpt/provider.toml index dffb70a9d..bc050b463 100644 --- a/providers/nano-gpt/provider.toml +++ b/providers/nano-gpt/provider.toml @@ -1,5 +1,15 @@ name = "NanoGPT" env = ["NANO_GPT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Chat variants are POST `/api/v1/chat/completions` (`reasoning`), +# `/api/v1legacy/chat/completions` (`reasoning_content`), and +# `/api/v1thinking/chat/completions` (reasoning merged into `content`). Requests +# accept `reasoning_effort` or `reasoning.effort`; exact values are `none`, +# `minimal`, `low`, `medium`, `high`, and `xhigh`, and `none` disables reasoning. +# `reasoning.exclude = true` only hides output. Separately, `:thinking`, legacy +# `-thinking`, and Anthropic `:` budgets are model-ID variants/aliases. +# https://docs.nano-gpt.com/api-reference/miscellaneous/extended-thinking +# https://docs.nano-gpt.com/api-reference/miscellaneous/model-suffixes +# (accessed 2026-06-25) api = "https://nano-gpt.com/api/v1" doc = "https://docs.nano-gpt.com" diff --git a/providers/nearai/provider.toml b/providers/nearai/provider.toml index bbb043615..0f027770a 100644 --- a/providers/nearai/provider.toml +++ b/providers/nearai/provider.toml @@ -1,5 +1,11 @@ name = "NEAR AI Cloud" npm = "@ai-sdk/openai-compatible" +# Gateway and direct TEE hosts use the same POST `/v1/chat/completions` API. +# Native GLM/Qwen controls use `chat_template_kwargs.enable_thinking = true|false`; +# GPT-OSS uses `reasoning_effort = low|medium|high` and cannot be disabled. +# Third-party routes pass through their provider-native reasoning controls. +# https://docs.near.ai/cloud/guides/openai-compatibility +# https://docs.near.ai/cloud/reasoning-models (accessed 2026-06-25) api = "https://cloud-api.near.ai/v1" env = ["NEARAI_API_KEY"] doc = "https://docs.near.ai/" diff --git a/providers/nebius/provider.toml b/providers/nebius/provider.toml index 4f61f36d2..2b00741de 100644 --- a/providers/nebius/provider.toml +++ b/providers/nebius/provider.toml @@ -1,5 +1,10 @@ name = "Nebius Token Factory" env = ["NEBIUS_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP is POST `/v1/chat/completions`; the generic request schema accepts +# `reasoning_effort = "low"|"medium"|"high"`. Model-native controls may also +# pass through because the request permits additional properties. +# https://docs.tokenfactory.nebius.com/api-reference/inference/create-chat-completion +# (accessed 2026-06-25) api = "https://api.tokenfactory.nebius.com/v1" doc = "https://docs.tokenfactory.nebius.com/" \ No newline at end of file diff --git a/providers/neon/provider.toml b/providers/neon/provider.toml index 2c51a10d8..759defa0d 100644 --- a/providers/neon/provider.toml +++ b/providers/neon/provider.toml @@ -1,5 +1,15 @@ name = "Neon" npm = "@ai-sdk/openai-compatible" +# Unified Chat is POST `/ai-gateway/mlflow/v1/chat/completions`. Native routes +# are POST `/ai-gateway/anthropic/v1/messages` (`thinking.type`, +# `thinking.budget_tokens`), POST `/ai-gateway/openai/v1/responses` +# (`reasoning.effort`), and POST +# `/ai-gateway/gemini/v1beta/models/{model}:generateContent` +# (`generationConfig.thinkingConfig`). +# https://neon.com/docs/ai-gateway/chat-completions +# https://neon.com/docs/ai-gateway/anthropic-messages +# https://neon.com/docs/ai-gateway/openai-responses +# https://neon.com/docs/ai-gateway/gemini (accessed 2026-06-25) api = "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/mlflow/v1" env = ["NEON_AI_GATEWAY_BASE_URL", "NEON_AI_GATEWAY_TOKEN"] doc = "https://neon.com/docs" diff --git a/providers/neuralwatt/provider.toml b/providers/neuralwatt/provider.toml index fea3334f6..6e207a1b2 100644 --- a/providers/neuralwatt/provider.toml +++ b/providers/neuralwatt/provider.toml @@ -1,5 +1,10 @@ name = "Neuralwatt" env = ["NEURALWATT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP is POST `/v1/chat/completions`. Reasoning models accept top-level +# `thinking_token_budget` (no bounds or disable sentinel documented). Native +# toggles use `chat_template_kwargs.enable_thinking = true|false`; GLM-5.2 also +# accepts `reasoning_effort`, with the model-specific normalization documented. +# https://portal.neuralwatt.com/docs/api/chat-completions (accessed 2026-06-25) api = "https://api.neuralwatt.com/v1" doc = "https://portal.neuralwatt.com/docs" diff --git a/providers/nova/provider.toml b/providers/nova/provider.toml index 878e9980c..bcee18c73 100644 --- a/providers/nova/provider.toml +++ b/providers/nova/provider.toml @@ -1,5 +1,11 @@ name = "Nova" env = ["NOVA_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Nova 2 Lite's native request uses `additionalModelRequestFields.reasoningConfig` +# with `type = "enabled"|"disabled"` (disabled by default) and, when enabled, +# `maxReasoningEffort = "low"|"medium"|"high"`. Nova 2 Pro has no documented +# selectable reasoning control. +# https://docs.aws.amazon.com/nova/latest/nova2-userguide/extended-thinking.html +# (accessed 2026-06-25) api = "https://api.nova.amazon.com/v1" doc = "https://nova.amazon.com/dev/documentation" diff --git a/providers/novita-ai/provider.toml b/providers/novita-ai/provider.toml index 05acfd40b..b7ebc2916 100644 --- a/providers/novita-ai/provider.toml +++ b/providers/novita-ai/provider.toml @@ -1,5 +1,11 @@ name = "NovitaAI" env = ["NOVITA_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST `/openai/v1/chat/completions` accepts top-level `enable_thinking = +# true|false` (default true), but documents it only for zai-org/glm-4.5 and +# deepseek/deepseek-v3.1, -v3.1-terminus, and -v3.2-exp. `separate_reasoning` +# is a distinct boolean documented only for deepseek/deepseek-r1-turbo. +# https://novita.ai/docs/api-reference/model-apis-llm-create-chat-completion doc = "https://novita.ai/docs/guides/introduction" api = "https://api.novita.ai/openai" diff --git a/providers/nvidia/models/deepseek-ai/deepseek-v4-flash.toml b/providers/nvidia/models/deepseek-ai/deepseek-v4-flash.toml index a1f18f787..130070ed5 100644 --- a/providers/nvidia/models/deepseek-ai/deepseek-v4-flash.toml +++ b/providers/nvidia/models/deepseek-ai/deepseek-v4-flash.toml @@ -1,5 +1,7 @@ base_model = "deepseek/deepseek-v4-flash" +# NIM Chat schema: `reasoning_effort = none|high|max`; `none` disables thinking. +# https://docs.api.nvidia.com/nim/reference/deepseek-ai-deepseek-v4-flash-infer reasoning_options = [{ type = "effort", values = ["none", "high", "max"] }] [interleaved] diff --git a/providers/nvidia/models/deepseek-ai/deepseek-v4-pro.toml b/providers/nvidia/models/deepseek-ai/deepseek-v4-pro.toml index 5aa053645..0fc901122 100644 --- a/providers/nvidia/models/deepseek-ai/deepseek-v4-pro.toml +++ b/providers/nvidia/models/deepseek-ai/deepseek-v4-pro.toml @@ -1,5 +1,7 @@ base_model = "deepseek/deepseek-v4-pro" +# NIM Chat schema: `reasoning_effort = none|high|max`; `none` disables thinking. +# https://docs.api.nvidia.com/nim/reference/deepseek-ai-deepseek-v4-pro-infer reasoning_options = [{ type = "effort", values = ["none", "high", "max"] }] [interleaved] diff --git a/providers/nvidia/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning.toml b/providers/nvidia/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning.toml index 00eb69c88..572c8334d 100644 --- a/providers/nvidia/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning.toml +++ b/providers/nvidia/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning.toml @@ -6,6 +6,10 @@ release_date = "2026-04-28" last_updated = "2026-04-28" attachment = true reasoning = true +# NIM Chat schema: `reasoning_budget` is -1..32768 (default 16384); -1 means +# unlimited/no budget enforcement, not reasoning off. `/think` and `/no_think` +# go in a system message; video and audio requests should use `/no_think`. +# https://docs.api.nvidia.com/nim/reference/nvidia-nemotron-3-nano-omni-30b-a3b-reasoning-infer reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = -1, max = 32768 }] temperature = true tool_call = true diff --git a/providers/nvidia/models/qwen/qwen3.5-122b-a10b.toml b/providers/nvidia/models/qwen/qwen3.5-122b-a10b.toml index 460c7a89b..41ba8a352 100644 --- a/providers/nvidia/models/qwen/qwen3.5-122b-a10b.toml +++ b/providers/nvidia/models/qwen/qwen3.5-122b-a10b.toml @@ -4,6 +4,8 @@ release_date = "2026-02-23" last_updated = "2026-02-23" attachment = true reasoning = true +# NIM Chat schema: `chat_template_kwargs.enable_thinking = true|false`. +# https://docs.api.nvidia.com/nim/reference/qwen-qwen3-5-122b-a10b-infer reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true diff --git a/providers/nvidia/models/qwen/qwen3.5-397b-a17b.toml b/providers/nvidia/models/qwen/qwen3.5-397b-a17b.toml index 7d1178abd..165646711 100644 --- a/providers/nvidia/models/qwen/qwen3.5-397b-a17b.toml +++ b/providers/nvidia/models/qwen/qwen3.5-397b-a17b.toml @@ -4,6 +4,8 @@ release_date = "2026-02-16" last_updated = "2026-02-16" attachment = true reasoning = true +# NIM Chat schema: `chat_template_kwargs.enable_thinking = true|false`. +# https://docs.api.nvidia.com/nim/reference/qwen-qwen3-5-397b-a17b-infer reasoning_options = [{ type = "toggle" }] temperature = true knowledge = "2026-01" diff --git a/providers/nvidia/provider.toml b/providers/nvidia/provider.toml index d1a4ec5dd..1fed7e206 100644 --- a/providers/nvidia/provider.toml +++ b/providers/nvidia/provider.toml @@ -1,5 +1,11 @@ name = "Nvidia" env = ["NVIDIA_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# POST `/v1/chat/completions` has a model-specific NIM request schema; there is +# no provider-wide reasoning field or enum. Examples include top-level +# `reasoning_effort`, `chat_template_kwargs.enable_thinking`, prompt `/think` +# and `/no_think`, and top-level `reasoning_budget`. Check the exact model NIM. +# https://docs.api.nvidia.com/nim/reference/llm-apis doc = "https://docs.api.nvidia.com/nim/" api = "https://integrate.api.nvidia.com/v1" \ No newline at end of file diff --git a/providers/ollama-cloud/provider.toml b/providers/ollama-cloud/provider.toml index 9b6853e29..0549d3db6 100644 --- a/providers/ollama-cloud/provider.toml +++ b/providers/ollama-cloud/provider.toml @@ -1,5 +1,13 @@ name = "Ollama Cloud" env = ["OLLAMA_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# Native POST `/api/chat` and `/api/generate` use `think = true|false` or a +# model-supported level. GPT-OSS accepts only low|medium|high; booleans are +# ignored and its trace cannot be disabled. Native output is `message.thinking` +# or `thinking`. OpenAI POST `/v1/chat/completions` instead accepts top-level +# `reasoning_effort` or `reasoning.effort` with high|medium|low|max|none. +# https://docs.ollama.com/capabilities/thinking +# https://docs.ollama.com/openai api = "https://ollama.com/v1" doc = "https://docs.ollama.com/cloud" diff --git a/providers/opencode-go/provider.toml b/providers/opencode-go/provider.toml index 4c1547c53..833a7cb00 100644 --- a/providers/opencode-go/provider.toml +++ b/providers/opencode-go/provider.toml @@ -1,5 +1,11 @@ name = "OpenCode Go" env = ["OPENCODE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP caveat (source accessed 2026-06-25): the Zen endpoint table does not +# document the `/zen/go/v1` API or any Go reasoning request fields, values, +# bounds, translation, or passthrough. The configured endpoint is therefore an +# undocumented surface; model reasoning options are not an HTTP contract cited +# by the public Zen page. +# https://opencode.ai/docs/zen#endpoints api = "https://opencode.ai/zen/go/v1" doc = "https://opencode.ai/docs/zen" diff --git a/providers/opencode/provider.toml b/providers/opencode/provider.toml index 72c8dc4e8..b052bdb7c 100644 --- a/providers/opencode/provider.toml +++ b/providers/opencode/provider.toml @@ -1,5 +1,12 @@ name = "OpenCode Zen" env = ["OPENCODE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP endpoint map (source accessed 2026-06-25): Zen uses POST +# `/zen/v1/responses` for OpenAI, `/zen/v1/messages` for Anthropic and Qwen, +# `/zen/v1/models/{model}` for Gemini, and `/zen/v1/chat/completions` for the +# listed compatible models. Zen does not document reasoning request fields, +# values, bounds, translation, or passthrough; upstream-native formats are not +# independently guaranteed by this endpoint table. +# https://opencode.ai/docs/zen#endpoints api = "https://opencode.ai/zen/v1" doc = "https://opencode.ai/docs/zen" diff --git a/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml b/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml index 495d3fab0..9d709129e 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.6-fast.toml @@ -1,4 +1,5 @@ -# OpenRouter maps these effort values through `verbosity`, not `reasoning.effort`. +# This route maps effort through top-level `verbosity`, not `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/parameters#verbosity (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1024, max = 127999 }] name = "Claude Opus 4.6 (Fast)" family = "claude-opus" diff --git a/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml b/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml index 87268bd09..d8c17e91b 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.7-fast.toml @@ -1,4 +1,5 @@ -# OpenRouter maps these effort values through `verbosity`, not `reasoning.effort`. +# This route maps effort through top-level `verbosity`, not `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/parameters#verbosity (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] name = "Claude Opus 4.7 (Fast)" family = "claude-opus" diff --git a/providers/openrouter/models/anthropic/claude-opus-4.8-fast.toml b/providers/openrouter/models/anthropic/claude-opus-4.8-fast.toml index d09ce4f95..c44b84506 100644 --- a/providers/openrouter/models/anthropic/claude-opus-4.8-fast.toml +++ b/providers/openrouter/models/anthropic/claude-opus-4.8-fast.toml @@ -1,4 +1,5 @@ -# OpenRouter maps these effort values through `verbosity`, not `reasoning.effort`. +# This route maps effort through top-level `verbosity`, not `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/parameters#verbosity (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] name = "Claude Opus 4.8 (Fast)" family = "claude-opus" diff --git a/providers/openrouter/models/~anthropic/claude-fable-latest.toml b/providers/openrouter/models/~anthropic/claude-fable-latest.toml index 0831520e5..e7ebdea82 100644 --- a/providers/openrouter/models/~anthropic/claude-fable-latest.toml +++ b/providers/openrouter/models/~anthropic/claude-fable-latest.toml @@ -1,4 +1,5 @@ -# OpenRouter maps these effort values through `verbosity`, not `reasoning.effort`. +# This route maps effort through top-level `verbosity`, not `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/parameters#verbosity (accessed 2026-06-25) reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] name = "Claude Fable Latest" family = "claude-fable" diff --git a/providers/openrouter/models/~anthropic/claude-opus-latest.toml b/providers/openrouter/models/~anthropic/claude-opus-latest.toml index 30d78e714..d495b320d 100644 --- a/providers/openrouter/models/~anthropic/claude-opus-latest.toml +++ b/providers/openrouter/models/~anthropic/claude-opus-latest.toml @@ -1,4 +1,5 @@ -# OpenRouter maps these effort values through `verbosity`, not `reasoning.effort`. +# This route maps effort through top-level `verbosity`, not `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/parameters#verbosity (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] name = "Claude Opus Latest" family = "claude-opus" diff --git a/providers/openrouter/models/~anthropic/claude-sonnet-latest.toml b/providers/openrouter/models/~anthropic/claude-sonnet-latest.toml index f056c8a43..122427ef6 100644 --- a/providers/openrouter/models/~anthropic/claude-sonnet-latest.toml +++ b/providers/openrouter/models/~anthropic/claude-sonnet-latest.toml @@ -1,4 +1,5 @@ -# OpenRouter maps these effort values through `verbosity`, not `reasoning.effort`. +# This route maps effort through top-level `verbosity`, not `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/parameters#verbosity (accessed 2026-06-25) reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1024, max = 127999 }] name = "Anthropic Claude Sonnet Latest" family = "claude-sonnet" diff --git a/providers/openrouter/provider.toml b/providers/openrouter/provider.toml index ea89f3dfd..34644f76c 100644 --- a/providers/openrouter/provider.toml +++ b/providers/openrouter/provider.toml @@ -1,5 +1,17 @@ name = "OpenRouter" env = ["OPENROUTER_API_KEY"] npm = "@openrouter/ai-sdk-provider" +# Raw Chat: `reasoning.enabled` toggles, `reasoning.effort` selects effort, +# `reasoning.max_tokens` sets a budget, and top-level `reasoning_effort` is an alias. +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens (accessed 2026-06-25) +# Raw Responses: POST `/responses` uses `reasoning.effort`. +# https://openrouter.ai/docs/api/reference/responses/reasoning (accessed 2026-06-25) +# Raw Anthropic Messages: POST `/messages` uses `thinking.type`, +# `thinking.budget_tokens`, and `output_config.effort`. +# https://openrouter.ai/docs/api/api-reference/anthropic-messages/create-messages (accessed 2026-06-25) +# Live `GET /api/v1/models` reasoning metadata is model-specific; use +# `provider.require_parameters = true` to exclude routes that ignore requested controls. +# https://openrouter.ai/api/v1/models (accessed 2026-06-25) +# https://openrouter.ai/docs/guides/routing/provider-selection#requiring-providers-to-support-all-parameters (accessed 2026-06-25) api = "https://openrouter.ai/api/v1" doc = "https://openrouter.ai/models" diff --git a/providers/orcarouter/provider.toml b/providers/orcarouter/provider.toml index 04026256e..03aa237ea 100644 --- a/providers/orcarouter/provider.toml +++ b/providers/orcarouter/provider.toml @@ -1,5 +1,16 @@ name = "OrcaRouter" env = ["ORCAROUTER_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# Chat POST `/v1/chat/completions` accepts `reasoning_effort = low|medium|high` +# plus model-specific minimal|max and translates it to the selected upstream. +# Messages POST `/v1/messages` accepts `thinking.type = enabled|disabled|adaptive` +# and `thinking.budget_tokens`. Gemini native POST +# `/v1beta/models/{model}:generateContent` passes through +# `generationConfig.thinkingConfig` (including `includeThoughts`, thinking +# level, or budget). DeepSeek reasoner's Chat effort is documented as a no-op. +# https://docs.orcarouter.ai/advanced/reasoning +# https://docs.orcarouter.ai/api-reference/messages/create-a-message +# https://docs.orcarouter.ai/native-formats/gemini api = "https://api.orcarouter.ai/v1" doc = "https://docs.orcarouter.ai" diff --git a/providers/ovhcloud/models/gpt-oss-120b.toml b/providers/ovhcloud/models/gpt-oss-120b.toml index 3a5e0a0df..160f434ff 100644 --- a/providers/ovhcloud/models/gpt-oss-120b.toml +++ b/providers/ovhcloud/models/gpt-oss-120b.toml @@ -3,6 +3,8 @@ release_date = "2025-08-28" last_updated = "2025-08-28" attachment = false reasoning = true +# Chat: `reasoning_effort = low|medium|high`; Responses: `reasoning.effort`. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/gpt-oss-120b/ reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] tool_call = true structured_output = true diff --git a/providers/ovhcloud/models/gpt-oss-20b.toml b/providers/ovhcloud/models/gpt-oss-20b.toml index e2c154a2e..06b7893c1 100644 --- a/providers/ovhcloud/models/gpt-oss-20b.toml +++ b/providers/ovhcloud/models/gpt-oss-20b.toml @@ -3,6 +3,8 @@ release_date = "2025-08-28" last_updated = "2025-08-28" attachment = false reasoning = true +# Chat: `reasoning_effort = low|medium|high`; Responses: `reasoning.effort`. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/gpt-oss-20b/ reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] tool_call = true structured_output = true diff --git a/providers/ovhcloud/models/qwen3-32b.toml b/providers/ovhcloud/models/qwen3-32b.toml index ffdc77096..21bfa9f7c 100644 --- a/providers/ovhcloud/models/qwen3-32b.toml +++ b/providers/ovhcloud/models/qwen3-32b.toml @@ -3,6 +3,8 @@ release_date = "2025-07-16" last_updated = "2025-07-16" attachment = false reasoning = true +# Put `/no_think` in prompt content to disable the model's default reasoning. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-32b/ reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true diff --git a/providers/ovhcloud/models/qwen3.5-397b-a17b.toml b/providers/ovhcloud/models/qwen3.5-397b-a17b.toml index 00ff86a11..8f5c004d4 100644 --- a/providers/ovhcloud/models/qwen3.5-397b-a17b.toml +++ b/providers/ovhcloud/models/qwen3.5-397b-a17b.toml @@ -3,6 +3,8 @@ release_date = "2026-05-18" last_updated = "2026-05-18" attachment = true reasoning = true +# Chat `reasoning_effort` supports none|low|medium|high; `none` disables it. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-5-397b/ reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/ovhcloud/models/qwen3.5-9b.toml b/providers/ovhcloud/models/qwen3.5-9b.toml index 4fe9ba0ea..23f82952d 100644 --- a/providers/ovhcloud/models/qwen3.5-9b.toml +++ b/providers/ovhcloud/models/qwen3.5-9b.toml @@ -3,6 +3,8 @@ release_date = "2026-04-22" last_updated = "2026-04-22" attachment = true reasoning = true +# Chat `reasoning_effort` supports none|low|medium|high; `none` disables it. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-5-9b/ reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/ovhcloud/models/qwen3.6-27b.toml b/providers/ovhcloud/models/qwen3.6-27b.toml index 83a24178d..59ee7a666 100644 --- a/providers/ovhcloud/models/qwen3.6-27b.toml +++ b/providers/ovhcloud/models/qwen3.6-27b.toml @@ -3,6 +3,8 @@ release_date = "2026-06-01" last_updated = "2026-06-01" attachment = true reasoning = true +# Chat `reasoning_effort` supports none|minimal|low|medium|high; `none` disables it. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-6-27b/ reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/ovhcloud/provider.toml b/providers/ovhcloud/provider.toml index f6a3efde7..48906b07e 100644 --- a/providers/ovhcloud/provider.toml +++ b/providers/ovhcloud/provider.toml @@ -1,5 +1,11 @@ name = "OVHcloud AI Endpoints" npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): OVH exposes both +# POST `/v1/chat/completions` and `/v1/responses`, but controls are model-specific. +# GPT-OSS uses Chat `reasoning_effort` or Responses `reasoning.effort`; Qwen3-32B +# uses `/no_think` in prompt content. See the model comments for exact enums. +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/gpt-oss-120b/ +# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-32b/ api = "https://oai.endpoints.kepler.ai.cloud.ovh.net/v1" env = ["OVHCLOUD_API_KEY"] doc = "https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog//" \ No newline at end of file diff --git a/providers/perplexity/models/sonar-deep-research.toml b/providers/perplexity/models/sonar-deep-research.toml index 2f1b9d673..a5df9459a 100644 --- a/providers/perplexity/models/sonar-deep-research.toml +++ b/providers/perplexity/models/sonar-deep-research.toml @@ -1,3 +1,10 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.perplexity.ai/v1/sonar +# JSON reasoning_effort: "minimal" | "low" | "medium" | "high". +# No toggle or reasoning-token budget is documented for this model. +# Sources: +# https://docs.perplexity.ai/api-reference/sonar-post +# https://docs.perplexity.ai/docs/sonar/models/sonar-deep-research name = "Perplexity Sonar Deep Research" release_date = "2025-02-01" last_updated = "2025-09-01" diff --git a/providers/perplexity/models/sonar-reasoning-pro.toml b/providers/perplexity/models/sonar-reasoning-pro.toml index 84596aeb8..a41bcce94 100644 --- a/providers/perplexity/models/sonar-reasoning-pro.toml +++ b/providers/perplexity/models/sonar-reasoning-pro.toml @@ -1,3 +1,10 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.perplexity.ai/v1/sonar +# JSON reasoning_effort: "minimal" | "low" | "medium" | "high". +# No toggle or reasoning-token budget is documented for this model. +# Sources: +# https://docs.perplexity.ai/api-reference/sonar-post +# https://docs.perplexity.ai/docs/sonar/models/sonar-reasoning-pro name = "Sonar Reasoning Pro" family = "sonar-reasoning" release_date = "2024-01-01" diff --git a/providers/perplexity/provider.toml b/providers/perplexity/provider.toml index fabba7246..b983d2bdc 100644 --- a/providers/perplexity/provider.toml +++ b/providers/perplexity/provider.toml @@ -1,4 +1,10 @@ name = "Perplexity" env = ["PERPLEXITY_API_KEY"] npm = "@ai-sdk/perplexity" +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.perplexity.ai/v1/sonar +# JSON reasoning_effort: "minimal" | "low" | "medium" | "high" | null. +# No raw HTTP reasoning toggle or reasoning-token budget is documented. +# Sources: +# https://docs.perplexity.ai/api-reference/sonar-post doc = "https://docs.perplexity.ai" diff --git a/providers/poe/provider.toml b/providers/poe/provider.toml index 17d03c091..b05ef90b2 100644 --- a/providers/poe/provider.toml +++ b/providers/poe/provider.toml @@ -1,5 +1,12 @@ name = "Poe" env = ["POE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (source accessed 2026-06-25): Responses POST +# `/v1/responses` accepts `reasoning.effort` and `reasoning.summary`. On Chat +# POST `/v1/chat/completions`, top-level `reasoning_effort` is ignored; pass bot +# controls such as `reasoning_effort` or `thinking_budget` through `extra_body`. +# Forwarding is best-effort, varies by bot, and unsupported fields are generally +# silently ignored, so acceptance alone does not prove an effective control. +# https://creator.poe.com/docs/external-applications/openai-compatible-api api = "https://api.poe.com/v1" doc= "https://creator.poe.com/docs/external-applications/openai-compatible-api" diff --git a/providers/poolside/models/poolside/laguna-m.1.toml b/providers/poolside/models/poolside/laguna-m.1.toml index 30eaee8ea..05e4085ba 100644 --- a/providers/poolside/models/poolside/laguna-m.1.toml +++ b/providers/poolside/models/poolside/laguna-m.1.toml @@ -3,6 +3,9 @@ release_date = "2026-04-28" last_updated = "2026-06-13" attachment = false reasoning = true +# Laguna-specific support is represented here; the public API schema supplies +# the exact `reasoning.effort` enum, while this model page only confirms API use. +# https://docs.poolside.ai/release-notes/laguna-m1 reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh"] }] temperature = true tool_call = true diff --git a/providers/poolside/models/poolside/laguna-xs.2.toml b/providers/poolside/models/poolside/laguna-xs.2.toml index 08ac2cab3..975f11b5c 100644 --- a/providers/poolside/models/poolside/laguna-xs.2.toml +++ b/providers/poolside/models/poolside/laguna-xs.2.toml @@ -3,6 +3,9 @@ release_date = "2026-04-28" last_updated = "2026-06-13" attachment = false reasoning = true +# Laguna-specific support is represented here; the public API schema supplies +# the exact `reasoning.effort` enum, while this model page only confirms API use. +# https://docs.poolside.ai/release-notes/laguna-xs2 reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh"] }] temperature = true tool_call = true diff --git a/providers/poolside/provider.toml b/providers/poolside/provider.toml index c7942028f..5bb0eb843 100644 --- a/providers/poolside/provider.toml +++ b/providers/poolside/provider.toml @@ -1,5 +1,13 @@ name = "Poolside" env = ["POOLSIDE_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): POST +# `/openai/v1/chat/completions` has generic `reasoning.effort` with +# none|minimal|low|medium|high|xhigh and allows additional properties. That +# schema describes the API surface, not support by every model; the published +# Laguna model pages establish OpenAI-compatible use but do not repeat the enum. +# https://docs.poolside.ai/openai-api/chat/create-chat-completion +# https://docs.poolside.ai/release-notes/laguna-m1 +# https://docs.poolside.ai/release-notes/laguna-xs2 api = "https://inference.poolside.ai/v1" doc = "https://platform.poolside.ai" diff --git a/providers/privatemode-ai/provider.toml b/providers/privatemode-ai/provider.toml index 4e43f7857..0c3e97cd7 100644 --- a/providers/privatemode-ai/provider.toml +++ b/providers/privatemode-ai/provider.toml @@ -1,5 +1,11 @@ name = "Privatemode AI" env = ["PRIVATEMODE_API_KEY", "PRIVATEMODE_ENDPOINT"] npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning caveat (source accessed 2026-06-25): POST +# `/v1/chat/completions` is served by the local Privatemode proxy. Additional +# OpenAI-style request parameters are passed through according to the active +# model server's capabilities; Privatemode defines no provider-wide reasoning +# enum or bounds. Thus gpt-oss `reasoning_effort` is model-server behavior. +# https://docs.privatemode.ai/api/chat-completions/#request-body api = "http://localhost:8080/v1" doc= "https://docs.privatemode.ai/api/overview" diff --git a/providers/qihang-ai/provider.toml b/providers/qihang-ai/provider.toml index 4d19fbe4a..4ee4fb309 100644 --- a/providers/qihang-ai/provider.toml +++ b/providers/qihang-ai/provider.toml @@ -2,4 +2,9 @@ name = "QiHang" npm = "@ai-sdk/openai-compatible" api = "https://api.qhaigc.net/v1" env = ["QIHANG_API_KEY"] +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions is documented as OpenAI compatible, but the public +# request reference does not document a reasoning control or accepted values. +# Source: +# https://www.qhaigc.net/docs/api-reference/chat/openai-style.md doc = "https://www.qhaigc.net/docs" diff --git a/providers/qiniu-ai/models/mimo-v2-flash.toml b/providers/qiniu-ai/models/mimo-v2-flash.toml index addb5a092..fb32c809c 100644 --- a/providers/qiniu-ai/models/mimo-v2-flash.toml +++ b/providers/qiniu-ai/models/mimo-v2-flash.toml @@ -1,4 +1,7 @@ base_model = "xiaomi/mimo-v2-flash" +# No model-specific Qiniu documentation maps the generic reasoning_effort or +# thinking fields to this catalog alias (accessed 2026-06-25). +# https://developer.qiniu.com/aitokenapi/13390/chat-completions reasoning_options = [] name = "Mimo-V2-Flash" diff --git a/providers/qiniu-ai/models/xiaomi/mimo-v2-flash.toml b/providers/qiniu-ai/models/xiaomi/mimo-v2-flash.toml index 4256782db..187e9f537 100644 --- a/providers/qiniu-ai/models/xiaomi/mimo-v2-flash.toml +++ b/providers/qiniu-ai/models/xiaomi/mimo-v2-flash.toml @@ -1,4 +1,7 @@ base_model = "xiaomi/mimo-v2-flash" +# No model-specific Qiniu documentation maps the generic reasoning_effort or +# thinking fields to this catalog ID (accessed 2026-06-25). +# https://developer.qiniu.com/aitokenapi/13390/chat-completions reasoning_options = [] name = "Xiaomi/Mimo-V2-Flash" diff --git a/providers/qiniu-ai/provider.toml b/providers/qiniu-ai/provider.toml index 75a2e60d1..7be8f1da0 100644 --- a/providers/qiniu-ai/provider.toml +++ b/providers/qiniu-ai/provider.toml @@ -2,4 +2,11 @@ name = "Qiniu" env = ["QINIU_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.qnaigc.com/v1" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions accepts top-level reasoning_effort = +# "low"|"medium"|"high", or thinking = {type = "enabled", +# budget_tokens = N}. The generic schema says model support varies; it does not +# map either field or a budget range to every catalog model. +# Source: +# https://developer.qiniu.com/aitokenapi/13390/chat-completions doc= "https://developer.qiniu.com/aitokenapi" diff --git a/providers/regolo-ai/models/gpt-oss-120b.toml b/providers/regolo-ai/models/gpt-oss-120b.toml index 39b2d1750..1dd52cb3e 100644 --- a/providers/regolo-ai/models/gpt-oss-120b.toml +++ b/providers/regolo-ai/models/gpt-oss-120b.toml @@ -4,6 +4,9 @@ release_date = "2025-08-05" last_updated = "2025-08-05" attachment = false reasoning = true +# Regolo documents reasoning_effort = "low"|"medium"|"high" for this model; +# thinking = true enables reasoning (accessed 2026-06-25). +# https://docs.regolo.ai/models/features/thinking/ reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/regolo-ai/models/gpt-oss-20b.toml b/providers/regolo-ai/models/gpt-oss-20b.toml index 00bc75fce..60dc8bbd4 100644 --- a/providers/regolo-ai/models/gpt-oss-20b.toml +++ b/providers/regolo-ai/models/gpt-oss-20b.toml @@ -4,6 +4,9 @@ release_date = "2026-03-01" last_updated = "2026-03-01" attachment = false reasoning = true +# Regolo documents these fields with gpt-oss-120b; it does not separately map +# them to gpt-oss-20b (accessed 2026-06-25). +# https://docs.regolo.ai/models/features/thinking/ reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/regolo-ai/provider.toml b/providers/regolo-ai/provider.toml index c0aab7eb4..568498b3a 100644 --- a/providers/regolo-ai/provider.toml +++ b/providers/regolo-ai/provider.toml @@ -1,5 +1,11 @@ name = "Regolo AI" env = ["REGOLO_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level thinking = true to enable thinking +# and reasoning_effort = "low"|"medium"|"high" (default "medium"). The +# public guide demonstrates GPT-OSS, but does not map controls to other models. +# Source: +# https://docs.regolo.ai/models/features/thinking/ doc = "https://docs.regolo.ai/" api = "https://api.regolo.ai/v1" diff --git a/providers/requesty/provider.toml b/providers/requesty/provider.toml index 053c769cc..5c89c2ff8 100644 --- a/providers/requesty/provider.toml +++ b/providers/requesty/provider.toml @@ -2,4 +2,14 @@ name = "Requesty" env = ["REQUESTY_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://router.requesty.ai/v1" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level reasoning_effort as an effort string +# or decimal budget string. OpenAI: low|medium|high pass through, max -> high, +# none|min -> low; budgets 0..1024/1025..8192/>=8193 map low/medium/high. +# Anthropic: none|min|low -> 1024, medium -> 8192, high -> 16384, max -> +# model max output minus 1. Vertex: none|min -> 0 for Flash/Flash Lite or 128 +# for Pro, low -> 1024, medium -> 8192, high -> 24576, max -> max output. +# Thus none is a gateway translation and does not always disable reasoning. +# Source: +# https://docs.requesty.ai/features/reasoning doc = "https://requesty.ai/solution/llm-routing/models" diff --git a/providers/routing-run/provider.toml b/providers/routing-run/provider.toml index fbb4aa6c3..b3e2c1a8a 100644 --- a/providers/routing-run/provider.toml +++ b/providers/routing-run/provider.toml @@ -2,4 +2,9 @@ name = "routing.run" env = ["ROUTING_RUN_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://ai.routing.sh/v1" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions may return message.reasoning_content, but the +# documented request body has no reasoning toggle, effort, or token budget. +# Source: +# https://docs.routing.run/api-reference/openai doc = "https://docs.routing.run/api-reference/models" diff --git a/providers/sap-ai-core/models/anthropic--claude-3.7-sonnet.toml b/providers/sap-ai-core/models/anthropic--claude-3.7-sonnet.toml index daf86b668..cf5a97d2b 100644 --- a/providers/sap-ai-core/models/anthropic--claude-3.7-sonnet.toml +++ b/providers/sap-ai-core/models/anthropic--claude-3.7-sonnet.toml @@ -4,6 +4,7 @@ release_date = "2025-02-24" last_updated = "2025-02-24" attachment = true reasoning = true +# Bedrock Invoke body: $.thinking = {"type":"enabled","budget_tokens":N} or {"type":"disabled"}, with N >= 1024 and N < $.max_tokens. Bedrock Converse nests that object at $.additionalModelRequestFields.thinking. https://docs.aws.amazon.com/bedrock/latest/userguide/conversation-inference.html#converse-additional-model-request-fields (accessed 2026-06-25) temperature = true tool_call = true knowledge = "2024-10-31" diff --git a/providers/sap-ai-core/models/anthropic--claude-4.6-opus.toml b/providers/sap-ai-core/models/anthropic--claude-4.6-opus.toml index d2d7f59f9..b50558360 100644 --- a/providers/sap-ai-core/models/anthropic--claude-4.6-opus.toml +++ b/providers/sap-ai-core/models/anthropic--claude-4.6-opus.toml @@ -4,6 +4,7 @@ release_date = "2026-02-05" last_updated = "2026-03-13" attachment = true reasoning = true +# Bedrock Claude 4.6 prefers $.thinking.type = "adaptive" plus $.output_config.effort = "low"|"medium"|"high"|"max"; manual enabled budget_tokens >= 1024 is deprecated. Converse prefixes both paths with $.additionalModelRequestFields. https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking (accessed 2026-06-25) temperature = true tool_call = true knowledge = "2025-05" diff --git a/providers/sap-ai-core/models/anthropic--claude-4.6-sonnet.toml b/providers/sap-ai-core/models/anthropic--claude-4.6-sonnet.toml index a82379430..ae1ef0e56 100644 --- a/providers/sap-ai-core/models/anthropic--claude-4.6-sonnet.toml +++ b/providers/sap-ai-core/models/anthropic--claude-4.6-sonnet.toml @@ -4,6 +4,7 @@ release_date = "2026-02-17" last_updated = "2026-03-13" attachment = true reasoning = true +# Bedrock Claude 4.6 prefers $.thinking.type = "adaptive" plus $.output_config.effort = "low"|"medium"|"high"; manual enabled budget_tokens >= 1024 is deprecated. Converse prefixes both paths with $.additionalModelRequestFields. https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking (accessed 2026-06-25) temperature = true tool_call = true knowledge = "2025-08" diff --git a/providers/sap-ai-core/models/anthropic--claude-4.7-opus.toml b/providers/sap-ai-core/models/anthropic--claude-4.7-opus.toml index 935b9eff1..15851e8ad 100644 --- a/providers/sap-ai-core/models/anthropic--claude-4.7-opus.toml +++ b/providers/sap-ai-core/models/anthropic--claude-4.7-opus.toml @@ -4,6 +4,7 @@ release_date = "2026-04-16" last_updated = "2026-04-16" attachment = true reasoning = true +# Bedrock Claude 4.7 uses $.thinking.type = "adaptive" and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected. Converse prefixes both paths with $.additionalModelRequestFields. https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking (accessed 2026-06-25) temperature = false tool_call = true knowledge = "2026-01-31" diff --git a/providers/sap-ai-core/models/gemini-2.5-flash-lite.toml b/providers/sap-ai-core/models/gemini-2.5-flash-lite.toml index a3ee3dbe7..173c6146e 100644 --- a/providers/sap-ai-core/models/gemini-2.5-flash-lite.toml +++ b/providers/sap-ai-core/models/gemini-2.5-flash-lite.toml @@ -4,6 +4,7 @@ release_date = "2025-06-17" last_updated = "2025-06-17" attachment = true reasoning = true +# Vertex Gemini adapter: $.generationConfig.thinkingConfig.thinkingBudget is 0 (off), -1 (dynamic), or 512..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true knowledge = "2025-01" tool_call = true diff --git a/providers/sap-ai-core/models/gemini-2.5-flash.toml b/providers/sap-ai-core/models/gemini-2.5-flash.toml index c6ae54d05..12a871684 100644 --- a/providers/sap-ai-core/models/gemini-2.5-flash.toml +++ b/providers/sap-ai-core/models/gemini-2.5-flash.toml @@ -4,6 +4,7 @@ release_date = "2025-04-17" last_updated = "2025-06-05" attachment = true reasoning = true +# Vertex Gemini adapter: $.generationConfig.thinkingConfig.thinkingBudget is 0 (off), -1 (dynamic), or 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true knowledge = "2025-01" tool_call = true diff --git a/providers/sap-ai-core/models/gemini-2.5-pro.toml b/providers/sap-ai-core/models/gemini-2.5-pro.toml index 194f8566a..f308683dc 100644 --- a/providers/sap-ai-core/models/gemini-2.5-pro.toml +++ b/providers/sap-ai-core/models/gemini-2.5-pro.toml @@ -4,6 +4,7 @@ release_date = "2025-03-25" last_updated = "2025-06-05" attachment = true reasoning = true +# Vertex Gemini adapter: $.generationConfig.thinkingConfig.thinkingBudget is -1 (dynamic) or 128..32768; 0/off is unsupported. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true knowledge = "2025-01" tool_call = true diff --git a/providers/sap-ai-core/models/gpt-5.4.toml b/providers/sap-ai-core/models/gpt-5.4.toml index 525a314d4..a6f9df141 100644 --- a/providers/sap-ai-core/models/gpt-5.4.toml +++ b/providers/sap-ai-core/models/gpt-5.4.toml @@ -4,6 +4,7 @@ release_date = "2026-03-05" last_updated = "2026-03-05" attachment = true reasoning = true +# Azure OpenAI deployment adapter: raw Chat uses top-level $.reasoning_effort = "none"|"low"|"medium"|"high"|"xhigh"; "none" is off. https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/reasoning (accessed 2026-06-25) temperature = false knowledge = "2025-08-31" tool_call = true diff --git a/providers/sap-ai-core/models/gpt-5.5.toml b/providers/sap-ai-core/models/gpt-5.5.toml index b5036652d..aa04bbafd 100644 --- a/providers/sap-ai-core/models/gpt-5.5.toml +++ b/providers/sap-ai-core/models/gpt-5.5.toml @@ -4,6 +4,7 @@ release_date = "2026-04-23" last_updated = "2026-04-23" attachment = true reasoning = true +# Azure OpenAI deployment adapter: raw Chat uses top-level $.reasoning_effort = "none"|"low"|"medium"|"high"|"xhigh"; "none" is off. https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/reasoning (accessed 2026-06-25) temperature = false knowledge = "2025-12-01" tool_call = true diff --git a/providers/sap-ai-core/models/gpt-5.toml b/providers/sap-ai-core/models/gpt-5.toml index ae468bcf6..d15245424 100644 --- a/providers/sap-ai-core/models/gpt-5.toml +++ b/providers/sap-ai-core/models/gpt-5.toml @@ -4,6 +4,7 @@ release_date = "2025-08-07" last_updated = "2025-08-07" attachment = true reasoning = true +# Azure OpenAI deployment adapter: raw Chat uses top-level $.reasoning_effort = "minimal"|"low"|"medium"|"high"; there is no explicit off value for original gpt-5. https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/reasoning (accessed 2026-06-25) temperature = false knowledge = "2024-09-30" tool_call = true diff --git a/providers/sap-ai-core/provider.toml b/providers/sap-ai-core/provider.toml index 06e3db694..ae8ade4f2 100644 --- a/providers/sap-ai-core/provider.toml +++ b/providers/sap-ai-core/provider.toml @@ -1,4 +1,5 @@ name = "SAP AI Core" env = ["AICORE_SERVICE_KEY"] npm = "@jerome-benoit/sap-ai-provider-v2" +# Raw orchestration v2 nests routed controls at $.config.modules.prompt_templating.model.params.; for example, reasoning_effort stays under params rather than at request root. https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/orchestration-workflow-v2 (accessed 2026-06-25) doc = "https://help.sap.com/docs/sap-ai-core" diff --git a/providers/sarvam/provider.toml b/providers/sarvam/provider.toml index fe24026e8..a4020a524 100644 --- a/providers/sarvam/provider.toml +++ b/providers/sarvam/provider.toml @@ -1,5 +1,12 @@ name = "Sarvam AI" env = ["SARVAM_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.sarvam.ai/v1/chat/completions accepts top-level +# reasoning_effort = "low"|"medium"|"high" (default "medium"). JSON null, +# not the string "null", disables reasoning for sarvam-30b and sarvam-105b. +# Sources: +# https://docs.sarvam.ai/api-reference-docs/chat/chat-completions +# https://docs.sarvam.ai/api-reference-docs/api-guides-tutorials/chat-completion/how-to/adjust-the-models-thinking-level doc = "https://docs.sarvam.ai/api-reference-docs/getting-started/models" api = "https://api.sarvam.ai/v1" diff --git a/providers/scaleway/models/gemma-3-27b-it.toml b/providers/scaleway/models/gemma-3-27b-it.toml index 8697d788d..afcfb4259 100644 --- a/providers/scaleway/models/gemma-3-27b-it.toml +++ b/providers/scaleway/models/gemma-3-27b-it.toml @@ -4,6 +4,9 @@ release_date = "2024-12-01" last_updated = "2026-03-17" attachment = true reasoning = true +# Scaleway does not list a reasoning_effort mapping for this model in its +# reasoning guide (accessed 2026-06-25). +# https://www.scaleway.com/en/docs/generative-apis/how-to/query-reasoning-models/ reasoning_options = [] temperature = true knowledge = "2024-12" diff --git a/providers/scaleway/models/gpt-oss-120b.toml b/providers/scaleway/models/gpt-oss-120b.toml index f9beb43e9..36231c6c2 100644 --- a/providers/scaleway/models/gpt-oss-120b.toml +++ b/providers/scaleway/models/gpt-oss-120b.toml @@ -4,6 +4,9 @@ release_date = "2024-01-01" last_updated = "2026-03-17" attachment = true reasoning = false +# Scaleway excludes gpt-oss-120b from reasoning_effort = "none"; documented +# effort values are "low"|"medium"|"high" (accessed 2026-06-25). +# https://www.scaleway.com/en/docs/generative-apis/how-to/query-reasoning-models/ reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] temperature = true tool_call = true diff --git a/providers/scaleway/models/qwen3-235b-a22b-instruct-2507.toml b/providers/scaleway/models/qwen3-235b-a22b-instruct-2507.toml index 0ee6096a8..b4b273c89 100644 --- a/providers/scaleway/models/qwen3-235b-a22b-instruct-2507.toml +++ b/providers/scaleway/models/qwen3-235b-a22b-instruct-2507.toml @@ -1,4 +1,7 @@ base_model = "alibaba/qwen3-235b-a22b" +# Scaleway does not list a reasoning_effort mapping for this model in its +# reasoning guide (accessed 2026-06-25). +# https://www.scaleway.com/en/docs/generative-apis/how-to/query-reasoning-models/ reasoning_options = [] name = "Qwen3 235B A22B Instruct 2507" release_date = "2025-07-01" diff --git a/providers/scaleway/provider.toml b/providers/scaleway/provider.toml index 8744e80d1..cb348143b 100644 --- a/providers/scaleway/provider.toml +++ b/providers/scaleway/provider.toml @@ -2,4 +2,11 @@ name = "Scaleway" npm = "@ai-sdk/openai-compatible" api = "https://api.scaleway.ai/v1" env = ["SCALEWAY_API_KEY"] +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level reasoning_effort; supported values +# vary by model. Omission enables reasoning. "none" disables it for most +# reasoning models, but not gpt-oss-120b. Legacy deepseek-r1-distill-llama-70b +# has no reasoning_effort and emits reasoning inside content tags. +# Source: +# https://www.scaleway.com/en/docs/generative-apis/how-to/query-reasoning-models/ doc = "https://www.scaleway.com/en/docs/generative-apis/" \ No newline at end of file diff --git a/providers/siliconflow-cn/models/deepseek-ai/DeepSeek-V4-Flash.toml b/providers/siliconflow-cn/models/deepseek-ai/DeepSeek-V4-Flash.toml index f89d3f029..cafd58c6d 100644 --- a/providers/siliconflow-cn/models/deepseek-ai/DeepSeek-V4-Flash.toml +++ b/providers/siliconflow-cn/models/deepseek-ai/DeepSeek-V4-Flash.toml @@ -1,4 +1,8 @@ base_model = "deepseek/deepseek-v4-flash" +# thinking_budget = 128..32768. This model alone also accepts +# reasoning_effort = "high"|"max"; low|medium map to high and xhigh maps to +# max (accessed 2026-06-25). +# https://docs.siliconflow.cn/cn/api-reference/chat-completions/chat-completions reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] [interleaved] diff --git a/providers/siliconflow-cn/models/tencent/Hunyuan-A13B-Instruct.toml b/providers/siliconflow-cn/models/tencent/Hunyuan-A13B-Instruct.toml index 06e065cdb..92a632f88 100644 --- a/providers/siliconflow-cn/models/tencent/Hunyuan-A13B-Instruct.toml +++ b/providers/siliconflow-cn/models/tencent/Hunyuan-A13B-Instruct.toml @@ -4,6 +4,9 @@ release_date = "2025-06-30" last_updated = "2025-11-25" attachment = false reasoning = true +# enable_thinking = true|false for this model; the China schema documents +# thinking_budget = 128..32768 only for "most" reasoning models. +# https://docs.siliconflow.cn/cn/api-reference/chat-completions/chat-completions reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true diff --git a/providers/siliconflow-cn/provider.toml b/providers/siliconflow-cn/provider.toml index c476cc4d3..93b2ca433 100644 --- a/providers/siliconflow-cn/provider.toml +++ b/providers/siliconflow-cn/provider.toml @@ -2,4 +2,12 @@ name = "SiliconFlow (China)" env = ["SILICONFLOW_CN_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.siliconflow.cn/v1" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level enable_thinking = true|false only +# for the schema's enumerated models, and thinking_budget = 128..32768 for +# most reasoning models. No zero/negative disable sentinel is documented. +# reasoning_effort is only for deepseek-ai/DeepSeek-V4-Flash: "high"|"max"; +# low|medium map to high and xhigh maps to max for compatibility. +# Source: +# https://docs.siliconflow.cn/cn/api-reference/chat-completions/chat-completions doc = "https://cloud.siliconflow.com/models" diff --git a/providers/siliconflow/models/tencent/Hunyuan-A13B-Instruct.toml b/providers/siliconflow/models/tencent/Hunyuan-A13B-Instruct.toml index 1c30920b0..88e8484ea 100644 --- a/providers/siliconflow/models/tencent/Hunyuan-A13B-Instruct.toml +++ b/providers/siliconflow/models/tencent/Hunyuan-A13B-Instruct.toml @@ -4,6 +4,8 @@ release_date = "2025-06-30" last_updated = "2025-11-25" attachment = false reasoning = true +# enable_thinking = true|false; thinking_budget = 128..32768. +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 128, max = 32_768 }] temperature = true tool_call = true diff --git a/providers/siliconflow/provider.toml b/providers/siliconflow/provider.toml index 7866dd196..ea04d31a9 100644 --- a/providers/siliconflow/provider.toml +++ b/providers/siliconflow/provider.toml @@ -2,4 +2,11 @@ name = "SiliconFlow" env = ["SILICONFLOW_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.siliconflow.com/v1" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level enable_thinking = true|false only +# for the schema's enumerated models (default true), and thinking_budget = +# 128..32768 for all reasoning models. No zero/negative disable sentinel or +# reasoning_effort field is documented by the global endpoint schema. +# Source: +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions doc = "https://cloud.siliconflow.com/models" \ No newline at end of file diff --git a/providers/snowflake-cortex/provider.toml b/providers/snowflake-cortex/provider.toml index 7340773e5..9c6dc3c3c 100644 --- a/providers/snowflake-cortex/provider.toml +++ b/providers/snowflake-cortex/provider.toml @@ -2,4 +2,15 @@ name = "Snowflake Cortex" npm = "@ai-sdk/openai-compatible" api = "https://${SNOWFLAKE_ACCOUNT}.snowflakecomputing.com/api/v2/cortex/v1" env = ["SNOWFLAKE_ACCOUNT", "SNOWFLAKE_CORTEX_PAT"] +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI-compatible POST /api/v2/cortex/v1/chat/completions: OpenAI models use +# top-level reasoning_effort = "none"|"minimal"|"low"|"medium"|"high". +# Claude models use reasoning.effort or reasoning.max_tokens; reasoning_effort +# is ignored, reasoning.effort is converted to max_tokens, and max_tokens is +# the native budget. Other model families ignore all three controls. +# Anthropic-compatible POST /api/v2/cortex/v1/messages (Claude only): adaptive +# thinking uses thinking.type = "adaptive" and output_config.effort = +# "low"|"medium"|"high"|"max" for claude-opus-4-6+ and claude-sonnet-4-6. +# Source: +# https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api doc = "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api" diff --git a/providers/stackit/provider.toml b/providers/stackit/provider.toml index 9ab06dd8e..a33d9cd97 100644 --- a/providers/stackit/provider.toml +++ b/providers/stackit/provider.toml @@ -1,5 +1,10 @@ name = "STACKIT" env = ["STACKIT_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions is documented for GPT-OSS, but STACKIT does not +# document a caller reasoning toggle, effort field, token budget, or bounds. +# Source: +# https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models doc = "https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models" api = "https://api.openai-compat.model-serving.eu01.onstackit.cloud/v1" diff --git a/providers/stepfun-ai/provider.toml b/providers/stepfun-ai/provider.toml index 1f764bdbd..45ccca758 100644 --- a/providers/stepfun-ai/provider.toml +++ b/providers/stepfun-ai/provider.toml @@ -1,5 +1,12 @@ name = "StepFun AI" env = ["STEPFUN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# Step Plan exposes POST /step_plan/v1/chat/completions with top-level +# `reasoning_effort` and POST /step_plan/v1/messages with +# `output_config.effort`. step-3.7-flash accepts low/medium/high; +# step-3.5-flash-2603 accepts low/high. No plan Responses endpoint is listed. +# Source: +# https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api doc = "https://platform.stepfun.ai/docs/en/step-plan/integrations/open-code" api = "https://api.stepfun.ai/step_plan/v1" diff --git a/providers/stepfun/models/step-3.5-flash-2603.toml b/providers/stepfun/models/step-3.5-flash-2603.toml index d513ffadf..94bf0cfb7 100644 --- a/providers/stepfun/models/step-3.5-flash-2603.toml +++ b/providers/stepfun/models/step-3.5-flash-2603.toml @@ -8,6 +8,9 @@ tool_call = true knowledge = "2025-01" open_weights = true +# Chat `reasoning_effort` and Messages `output_config.effort` accept low/high +# (accessed 2026-06-25). +# https://platform.stepfun.com/docs/zh/guides/models/step-3.5-flash [[reasoning_options]] type = "effort" values = ["low", "high"] diff --git a/providers/stepfun/models/step-3.7-flash.toml b/providers/stepfun/models/step-3.7-flash.toml index 58f1213dd..245f30c1d 100644 --- a/providers/stepfun/models/step-3.7-flash.toml +++ b/providers/stepfun/models/step-3.7-flash.toml @@ -1,5 +1,8 @@ base_model = "stepfun/step-3.7-flash" +# Chat `reasoning_effort`, Messages `output_config.effort`, and Responses +# `reasoning.effort` accept low/medium/high (accessed 2026-06-25). +# https://platform.stepfun.com/docs/zh/guides/models/step-3.7-flash [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/stepfun/provider.toml b/providers/stepfun/provider.toml index e2116bb2b..1dba3c839 100644 --- a/providers/stepfun/provider.toml +++ b/providers/stepfun/provider.toml @@ -1,5 +1,17 @@ name = "StepFun" env = ["STEPFUN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level `reasoning_effort`; POST /v1/messages +# uses `output_config.effort`; POST /v1/responses uses `reasoning.effort`. +# Values are low/medium/high for step-3.7-flash; step-3.5-flash-2603 accepts +# low/high. Responses supports only step-3.7-flash. Chat returns reasoning at +# `choices[].message.reasoning` or streamed `choices[].delta.reasoning`; +# `reasoning_format` is general (default) or deepseek-style, the latter using +# `reasoning_content`. Responses streams `response.reasoning_text.delta`. +# Sources: +# https://platform.stepfun.com/docs/zh/api-reference/chat/chat-completion-create +# https://platform.stepfun.com/docs/zh/api-reference/chat/messages-create +# https://platform.stepfun.com/docs/zh/api-reference/responses/responses-create doc = "https://platform.stepfun.com/docs/zh/overview/concept" api = "https://api.stepfun.com/v1" diff --git a/providers/submodel/provider.toml b/providers/submodel/provider.toml index b0f75dbf3..b0013208a 100644 --- a/providers/submodel/provider.toml +++ b/providers/submodel/provider.toml @@ -1,5 +1,12 @@ name = "submodel" env = ["SUBMODEL_INSTAGEN_ACCESS_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# InstaGen documents POST /v1/chat/completions but no caller reasoning toggle, +# effort field, token budget, values, bounds, sentinel, or reasoning response +# path. Its model catalog labels models as reasoning/general only. +# Sources: +# https://submodel.gitbook.io/docs/instagen/api-reference +# https://submodel.gitbook.io/docs/instagen/overview-1/available-models api = "https://llm.submodel.ai/v1" doc = "https://submodel.gitbook.io" diff --git a/providers/synthetic/provider.toml b/providers/synthetic/provider.toml index 025f5ce3c..23f526293 100644 --- a/providers/synthetic/provider.toml +++ b/providers/synthetic/provider.toml @@ -1,5 +1,13 @@ name = "Synthetic" env = ["SYNTHETIC_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP metadata (accessed 2026-06-25): +# GET https://api.synthetic.new/openai/v1/models is the live model surface. +# Entries expose `supported_features` (including `reasoning`), context/output +# limits, modalities, pricing, and serving provider, but no reasoning request +# field, accepted values, budget bounds, or disable sentinel. +# Sources: +# https://api.synthetic.new/openai/v1/models +# https://synthetic.new/pricing doc = "https://synthetic.new/pricing" api = "https://api.synthetic.new/openai/v1" diff --git a/providers/tencent-coding-plan/models/glm-5.toml b/providers/tencent-coding-plan/models/glm-5.toml index 21def00b3..85fa49c97 100644 --- a/providers/tencent-coding-plan/models/glm-5.toml +++ b/providers/tencent-coding-plan/models/glm-5.toml @@ -4,6 +4,9 @@ release_date = "2026-02-11" last_updated = "2026-02-11" attachment = false reasoning = true +# Top-level `thinking.type` accepts enabled/disabled and defaults enabled +# (accessed 2026-06-25). +# https://cloud.tencent.com/document/product/1823/132061 reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true diff --git a/providers/tencent-coding-plan/models/kimi-k2.5.toml b/providers/tencent-coding-plan/models/kimi-k2.5.toml index 7cbef6a6f..9573b6077 100644 --- a/providers/tencent-coding-plan/models/kimi-k2.5.toml +++ b/providers/tencent-coding-plan/models/kimi-k2.5.toml @@ -4,6 +4,9 @@ release_date = "2026-01-27" last_updated = "2026-01-27" attachment = true reasoning = true +# Top-level `thinking.type` accepts enabled/disabled and defaults enabled +# (accessed 2026-06-25). +# https://cloud.tencent.com/document/product/1823/132232 reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true diff --git a/providers/tencent-coding-plan/provider.toml b/providers/tencent-coding-plan/provider.toml index 460a70b7e..9505acdd7 100644 --- a/providers/tencent-coding-plan/provider.toml +++ b/providers/tencent-coding-plan/provider.toml @@ -1,5 +1,14 @@ name = "Tencent Coding Plan (China)" env = ["TENCENT_CODING_PLAN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# OpenAI plan endpoint: POST /coding/v3/chat/completions. Top-level +# `thinking.type` is enabled or disabled; reasoning is returned at +# `choices[].message.reasoning_content` or streamed `choices[].delta.reasoning_content`. +# GLM-5 and Kimi-K2.5 default enabled and allow both values. Hunyuan thinking, +# Hunyuan-T1, and MiniMax-M2.5 are always-on; no plan effort control is listed. +# Sources: +# https://cloud.tencent.com/document/product/1823/130092 +# https://cloud.tencent.com/document/product/1823/131208 doc = "https://cloud.tencent.com/document/product/1772/128947" api = "https://api.lkeap.cloud.tencent.com/coding/v3" diff --git a/providers/tencent-tokenhub/models/hy3-preview.toml b/providers/tencent-tokenhub/models/hy3-preview.toml index 6d53ebcc2..6545493a1 100644 --- a/providers/tencent-tokenhub/models/hy3-preview.toml +++ b/providers/tencent-tokenhub/models/hy3-preview.toml @@ -8,6 +8,9 @@ tool_call = true temperature = true open_weights = true +# `thinking.type` accepts enabled/disabled (default disabled), while top-level +# `reasoning_effort` accepts low/medium/high (default low), accessed 2026-06-25. +# https://cloud.tencent.com/document/product/1823/131208 [cost] input = 0 output = 0 diff --git a/providers/tencent-tokenhub/provider.toml b/providers/tencent-tokenhub/provider.toml index 745d8ffc4..45d477d82 100644 --- a/providers/tencent-tokenhub/provider.toml +++ b/providers/tencent-tokenhub/provider.toml @@ -1,5 +1,13 @@ name = "Tencent TokenHub" env = ["TENCENT_TOKENHUB_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions uses top-level `thinking.type`: enabled or disabled, +# and top-level `reasoning_effort`: low, medium, or high. Reasoning is returned +# at `choices[].message.reasoning_content` or streamed at +# `choices[].delta.reasoning_content`. Model defaults and support differ. +# Sources: +# https://cloud.tencent.com/document/product/1823/130079 +# https://cloud.tencent.com/document/product/1823/131208 doc = "https://cloud.tencent.com/document/product/1823/130050" api = "https://tokenhub.tencentmaas.com/v1" diff --git a/providers/the-grid-ai/provider.toml b/providers/the-grid-ai/provider.toml index 19515ed29..ed32510fe 100644 --- a/providers/the-grid-ai/provider.toml +++ b/providers/the-grid-ai/provider.toml @@ -1,5 +1,15 @@ name = "The Grid AI" env = ["THEGRIDAI_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.thegrid.ai/v1/chat/completions and beta POST +# https://messages-beta.api.thegrid.ai/v1/messages route by instrument. The +# docs say callers choose reasoning depth through the instrument tier, not a +# dedicated thinking configuration; Messages has no `thinking` request field. +# The generic Chat schema lists `reasoning_effort`, but no instrument-specific +# effect is documented, so no caller reasoning control is verified here. +# Sources: +# https://thegrid.ai/docs/api-reference/consumption-api +# https://thegrid.ai/docs/instrument-specifications/current-instruments api = "https://api.thegrid.ai/v1" doc = "https://thegrid.ai/docs" diff --git a/providers/togetherai/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/togetherai/models/MiniMaxAI/MiniMax-M2.5.toml index bf8a46b67..70e5178e8 100644 --- a/providers/togetherai/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/togetherai/models/MiniMaxAI/MiniMax-M2.5.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# The current reasoning model table does not list M2.5 or document controls. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "MiniMax-M2.5" family = "minimax" release_date = "2026-02-12" diff --git a/providers/togetherai/models/MiniMaxAI/MiniMax-M2.7.toml b/providers/togetherai/models/MiniMaxAI/MiniMax-M2.7.toml index 94dcb5060..763c84c9e 100644 --- a/providers/togetherai/models/MiniMaxAI/MiniMax-M2.7.toml +++ b/providers/togetherai/models/MiniMaxAI/MiniMax-M2.7.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# Reasoning-only model; no toggle, effort, or reasoning-token budget is supported. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "MiniMax-M2.7" family = "minimax" release_date = "2026-03-18" diff --git a/providers/togetherai/models/Qwen/Qwen3.5-397B-A17B.toml b/providers/togetherai/models/Qwen/Qwen3.5-397B-A17B.toml index c75fbb195..427d3c747 100644 --- a/providers/togetherai/models/Qwen/Qwen3.5-397B-A17B.toml +++ b/providers/togetherai/models/Qwen/Qwen3.5-397B-A17B.toml @@ -1,3 +1,8 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false (thinking is on by default). +# Compatibility: chat_template_kwargs.thinking or .enable_thinking: boolean. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#enable-and-disable-reasoning name = "Qwen3.5 397B A17B" family = "qwen" release_date = "2026-02-16" diff --git a/providers/togetherai/models/Qwen/Qwen3.5-9B.toml b/providers/togetherai/models/Qwen/Qwen3.5-9B.toml index 6e01d3b74..e4422bbb8 100644 --- a/providers/togetherai/models/Qwen/Qwen3.5-9B.toml +++ b/providers/togetherai/models/Qwen/Qwen3.5-9B.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false (thinking is on by default). +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "Qwen3.5 9B" family = "qwen" release_date = "2026-03-03" diff --git a/providers/togetherai/models/Qwen/Qwen3.6-Plus.toml b/providers/togetherai/models/Qwen/Qwen3.6-Plus.toml index 2ec61c0bc..261ff13e1 100644 --- a/providers/togetherai/models/Qwen/Qwen3.6-Plus.toml +++ b/providers/togetherai/models/Qwen/Qwen3.6-Plus.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false (thinking is on by default). +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "Qwen3.6 Plus" family = "qwen" release_date = "2026-04-30" diff --git a/providers/togetherai/models/deepcogito/cogito-v2-1-671b.toml b/providers/togetherai/models/deepcogito/cogito-v2-1-671b.toml index eb7734feb..5208b9d99 100644 --- a/providers/togetherai/models/deepcogito/cogito-v2-1-671b.toml +++ b/providers/togetherai/models/deepcogito/cogito-v2-1-671b.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false (thinking is on by default). +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "Cogito v2.1 671B" family = "cogito" release_date = "2025-11-13" diff --git a/providers/togetherai/models/deepseek-ai/DeepSeek-R1.toml b/providers/togetherai/models/deepseek-ai/DeepSeek-R1.toml index b74387c95..5c7bdfd83 100644 --- a/providers/togetherai/models/deepseek-ai/DeepSeek-R1.toml +++ b/providers/togetherai/models/deepseek-ai/DeepSeek-R1.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions (dedicated endpoint model). +# Reasoning is returned in content within tags; no control is documented. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#think-tags-in-content name = "DeepSeek-R1" family = "deepseek-thinking" release_date = "2025-01-20" diff --git a/providers/togetherai/models/deepseek-ai/DeepSeek-V3-1.toml b/providers/togetherai/models/deepseek-ai/DeepSeek-V3-1.toml index 33ab302d4..3f675afcb 100644 --- a/providers/togetherai/models/deepseek-ai/DeepSeek-V3-1.toml +++ b/providers/togetherai/models/deepseek-ai/DeepSeek-V3-1.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions (dedicated endpoint model). +# JSON reasoning.enabled: true | false; function calling requires false. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#enable-and-disable-reasoning name = "DeepSeek V3.1" family = "deepseek" release_date = "2025-08-21" diff --git a/providers/togetherai/models/deepseek-ai/DeepSeek-V4-Pro.toml b/providers/togetherai/models/deepseek-ai/DeepSeek-V4-Pro.toml index d56e4dff8..627ebd95e 100644 --- a/providers/togetherai/models/deepseek-ai/DeepSeek-V4-Pro.toml +++ b/providers/togetherai/models/deepseek-ai/DeepSeek-V4-Pro.toml @@ -1,3 +1,8 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false; reasoning_effort: "high" | "max". +# "low"/"medium" map to "high"; "xhigh" maps to "max". +# Sources: https://docs.together.ai/docs/deepseek-v4-quickstart#reasoning-effort name = "DeepSeek V4 Pro" family = "deepseek" release_date = "2026-04-24" diff --git a/providers/togetherai/models/google/gemma-4-31B-it.toml b/providers/togetherai/models/google/gemma-4-31B-it.toml index 86b0e5cab..e1a98483d 100644 --- a/providers/togetherai/models/google/gemma-4-31B-it.toml +++ b/providers/togetherai/models/google/gemma-4-31B-it.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# Together's reasoning guide does not document toggle, effort, or budget controls. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "Gemma 4 31B Instruct" family = "gemma" release_date = "2026-04-07" diff --git a/providers/togetherai/models/moonshotai/Kimi-K2.5.toml b/providers/togetherai/models/moonshotai/Kimi-K2.5.toml index d83de5bd8..962158305 100644 --- a/providers/togetherai/models/moonshotai/Kimi-K2.5.toml +++ b/providers/togetherai/models/moonshotai/Kimi-K2.5.toml @@ -1,3 +1,8 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# Together's current reasoning model table does not list Kimi K2.5; raw HTTP +# reasoning.enabled: true | false acceptance remains unresolved. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "Kimi K2.5" family = "kimi-k2" release_date = "2026-01-27" diff --git a/providers/togetherai/models/moonshotai/Kimi-K2.6.toml b/providers/togetherai/models/moonshotai/Kimi-K2.6.toml index 21e6a7709..54e4cb99f 100644 --- a/providers/togetherai/models/moonshotai/Kimi-K2.6.toml +++ b/providers/togetherai/models/moonshotai/Kimi-K2.6.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false (thinking is on by default). +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#enable-and-disable-reasoning name = "Kimi K2.6" family = "kimi-k2" release_date = "2026-04-21" diff --git a/providers/togetherai/models/nvidia/nemotron-3-ultra-550b-a55b.toml b/providers/togetherai/models/nvidia/nemotron-3-ultra-550b-a55b.toml index 24d04529a..27a691321 100644 --- a/providers/togetherai/models/nvidia/nemotron-3-ultra-550b-a55b.toml +++ b/providers/togetherai/models/nvidia/nemotron-3-ultra-550b-a55b.toml @@ -1,3 +1,8 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false. Medium effort instead uses +# chat_template_kwargs.medium_effort: true; otherwise effort defaults to high. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#reasoning-effort name = "Nemotron 3 Ultra 550B A55B" base_model = "nvidia/nemotron-3-ultra-550b-a55b" family = "nemotron" diff --git a/providers/togetherai/models/openai/gpt-oss-120b.toml b/providers/togetherai/models/openai/gpt-oss-120b.toml index fd436daf5..24ed50007 100644 --- a/providers/togetherai/models/openai/gpt-oss-120b.toml +++ b/providers/togetherai/models/openai/gpt-oss-120b.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning_effort: "low" | "medium" | "high". +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#reasoning-effort name = "GPT OSS 120B" family = "gpt-oss" release_date = "2025-08-05" diff --git a/providers/togetherai/models/openai/gpt-oss-20b.toml b/providers/togetherai/models/openai/gpt-oss-20b.toml index 0daef1a57..320ddb391 100644 --- a/providers/togetherai/models/openai/gpt-oss-20b.toml +++ b/providers/togetherai/models/openai/gpt-oss-20b.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning_effort: "low" | "medium" | "high". +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#reasoning-effort name = "GPT OSS 20B" family = "gpt-oss" release_date = "2025-08-05" diff --git a/providers/togetherai/models/pearl-ai/gemma-4-31b-it.toml b/providers/togetherai/models/pearl-ai/gemma-4-31b-it.toml index 3704e25b7..2f95d96a3 100644 --- a/providers/togetherai/models/pearl-ai/gemma-4-31b-it.toml +++ b/providers/togetherai/models/pearl-ai/gemma-4-31b-it.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# Together's reasoning guide does not document toggle, effort, or budget controls. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "Pearl AI Gemma 4 31B Instruct" family = "gemma" release_date = "2026-04-07" diff --git a/providers/togetherai/models/zai-org/GLM-5.1.toml b/providers/togetherai/models/zai-org/GLM-5.1.toml index 828179350..49c19465f 100644 --- a/providers/togetherai/models/zai-org/GLM-5.1.toml +++ b/providers/togetherai/models/zai-org/GLM-5.1.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false (thinking is on by default). +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#supported-models name = "GLM-5.1" family = "glm" release_date = "2026-04-07" diff --git a/providers/togetherai/models/zai-org/GLM-5.toml b/providers/togetherai/models/zai-org/GLM-5.toml index 027c5cb36..399a75a3f 100644 --- a/providers/togetherai/models/zai-org/GLM-5.toml +++ b/providers/togetherai/models/zai-org/GLM-5.toml @@ -1,3 +1,7 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false; thinking is on by default. +# Sources: https://docs.together.ai/docs/inference/chat/reasoning#enable-and-disable-reasoning name = "GLM-5" base_model = "zhipuai/glm-5" family = "glm" diff --git a/providers/togetherai/provider.toml b/providers/togetherai/provider.toml index 6b1143145..c005dd999 100644 --- a/providers/togetherai/provider.toml +++ b/providers/togetherai/provider.toml @@ -1,4 +1,13 @@ name = "Together AI" env = ["TOGETHER_API_KEY"] npm = "@ai-sdk/togetherai" +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://api.together.ai/v1/chat/completions +# JSON reasoning.enabled: true | false; reasoning_effort: "low" | "medium" | +# "high" in the generic schema. Compatibility form: +# chat_template_kwargs.thinking or .enable_thinking: true | false. +# No reasoning-token budget is documented; max_tokens caps total output. +# Sources: +# https://docs.together.ai/reference/chat-completions +# https://docs.together.ai/docs/inference/chat/reasoning doc = "https://docs.together.ai/docs/serverless-models" diff --git a/providers/umans-ai-coding-plan/models/umans-coder.toml b/providers/umans-ai-coding-plan/models/umans-coder.toml index f6b2ccee3..2d6c1c8b2 100644 --- a/providers/umans-ai-coding-plan/models/umans-coder.toml +++ b/providers/umans-ai-coding-plan/models/umans-coder.toml @@ -1,6 +1,9 @@ base_model = "moonshotai/kimi-k2.7-code" name = "Umans Coder" temperature = false +# This alias currently routes to Kimi K2.7 Code: reasoning is always on +# (`can_disable=false`) with no effort levels (accessed 2026-06-25). +# https://api.code.umans.ai/v1/models/info reasoning_options = [] [interleaved] diff --git a/providers/umans-ai-coding-plan/models/umans-flash.toml b/providers/umans-ai-coding-plan/models/umans-flash.toml index d2e5fdc2b..cdb8bdb18 100644 --- a/providers/umans-ai-coding-plan/models/umans-flash.toml +++ b/providers/umans-ai-coding-plan/models/umans-flash.toml @@ -1,6 +1,9 @@ base_model = "alibaba/qwen3.6-35b-a3b" name = "Umans Flash" temperature = false +# Live metadata: levels none/low/medium/high, default medium, and +# `can_disable=true` (accessed 2026-06-25). +# https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] [interleaved] diff --git a/providers/umans-ai-coding-plan/models/umans-glm-5.1.toml b/providers/umans-ai-coding-plan/models/umans-glm-5.1.toml index ff343e1e7..fb4e76dd6 100644 --- a/providers/umans-ai-coding-plan/models/umans-glm-5.1.toml +++ b/providers/umans-ai-coding-plan/models/umans-glm-5.1.toml @@ -1,5 +1,7 @@ base_model = "zhipuai/glm-5.1" name = "GLM 5.1" +# Live metadata: levels none/medium, default medium, and `can_disable=true` +# (accessed 2026-06-25): https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }] [interleaved] diff --git a/providers/umans-ai-coding-plan/models/umans-glm-5.2.toml b/providers/umans-ai-coding-plan/models/umans-glm-5.2.toml index afeb12b80..ccc697fa0 100644 --- a/providers/umans-ai-coding-plan/models/umans-glm-5.2.toml +++ b/providers/umans-ai-coding-plan/models/umans-glm-5.2.toml @@ -1,5 +1,7 @@ base_model = "zhipuai/glm-5.2" name = "GLM 5.2" +# Live metadata: levels none/high/max, default high, and `can_disable=true` +# (accessed 2026-06-25): https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] [interleaved] diff --git a/providers/umans-ai-coding-plan/models/umans-kimi-k2.7.toml b/providers/umans-ai-coding-plan/models/umans-kimi-k2.7.toml index 6add5058b..43ef003cf 100644 --- a/providers/umans-ai-coding-plan/models/umans-kimi-k2.7.toml +++ b/providers/umans-ai-coding-plan/models/umans-kimi-k2.7.toml @@ -1,5 +1,7 @@ base_model = "moonshotai/kimi-k2.7-code" name = "Kimi K2.7 Code" +# Live metadata says reasoning is always on (`can_disable=false`) with no effort +# levels (accessed 2026-06-25): https://api.code.umans.ai/v1/models/info reasoning_options = [] [interleaved] diff --git a/providers/umans-ai-coding-plan/models/umans-qwen3.6-35b-a3b.toml b/providers/umans-ai-coding-plan/models/umans-qwen3.6-35b-a3b.toml index 8f91ed4fb..90268cef2 100644 --- a/providers/umans-ai-coding-plan/models/umans-qwen3.6-35b-a3b.toml +++ b/providers/umans-ai-coding-plan/models/umans-qwen3.6-35b-a3b.toml @@ -1,6 +1,9 @@ base_model = "alibaba/qwen3.6-35b-a3b" name = "Qwen3.6 35B A3B" temperature = false +# Live metadata: levels none/low/medium/high, default medium, and +# `can_disable=true` (accessed 2026-06-25). +# https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] [interleaved] diff --git a/providers/umans-ai-coding-plan/provider.toml b/providers/umans-ai-coding-plan/provider.toml index 54e14f68d..c50341721 100644 --- a/providers/umans-ai-coding-plan/provider.toml +++ b/providers/umans-ai-coding-plan/provider.toml @@ -1,5 +1,16 @@ name = "Umans AI Coding Plan" env = ["UMANS_AI_CODING_PLAN_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions accepts top-level `reasoning_effort`; POST +# /v1/messages accepts `thinking.type` and `thinking.budget_tokens`. Either +# control is accepted on either endpoint; if both are sent, `thinking` wins. +# `thinking.type` is enabled/disabled; effort maps none/low/medium/high, with +# other values such as minimal mapped to the nearest level. A budget must be +# below max output; it is reduced to fit, and too little room disables thinking. +# No fixed numeric budget bounds are published. Chat returns +# `reasoning_content`; Messages returns thinking blocks/thinking_delta events. +# Live model metadata: GET https://api.code.umans.ai/v1/models/info +# Source: https://app.umans.ai/offers/code/docs#reasoning--extended-thinking doc = "https://app.umans.ai/offers/code/docs" api = "https://api.code.umans.ai/v1" diff --git a/providers/umans-ai/models/umans-coder.toml b/providers/umans-ai/models/umans-coder.toml index 58167e198..7bec758f0 100644 --- a/providers/umans-ai/models/umans-coder.toml +++ b/providers/umans-ai/models/umans-coder.toml @@ -1,6 +1,9 @@ base_model = "moonshotai/kimi-k2.7-code" name = "Umans Coder" temperature = false +# This alias currently routes to Kimi K2.7 Code: reasoning is always on +# (`can_disable=false`) with no effort levels (accessed 2026-06-25). +# https://api.code.umans.ai/v1/models/info reasoning_options = [] [interleaved] diff --git a/providers/umans-ai/models/umans-flash.toml b/providers/umans-ai/models/umans-flash.toml index 5692ae7ad..718532758 100644 --- a/providers/umans-ai/models/umans-flash.toml +++ b/providers/umans-ai/models/umans-flash.toml @@ -1,6 +1,9 @@ base_model = "alibaba/qwen3.6-35b-a3b" name = "Umans Flash" temperature = false +# Live metadata: levels none/low/medium/high, default medium, and +# `can_disable=true` (accessed 2026-06-25). +# https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] [interleaved] diff --git a/providers/umans-ai/models/umans-glm-5.1.toml b/providers/umans-ai/models/umans-glm-5.1.toml index 35efd66eb..ec7037f73 100644 --- a/providers/umans-ai/models/umans-glm-5.1.toml +++ b/providers/umans-ai/models/umans-glm-5.1.toml @@ -1,5 +1,7 @@ base_model = "zhipuai/glm-5.1" name = "GLM 5.1" +# Live metadata: levels none/medium, default medium, and `can_disable=true` +# (accessed 2026-06-25): https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }] [interleaved] diff --git a/providers/umans-ai/models/umans-glm-5.2.toml b/providers/umans-ai/models/umans-glm-5.2.toml index b6942b8dd..9e2184d38 100644 --- a/providers/umans-ai/models/umans-glm-5.2.toml +++ b/providers/umans-ai/models/umans-glm-5.2.toml @@ -1,5 +1,7 @@ base_model = "zhipuai/glm-5.2" name = "GLM 5.2" +# Live metadata: levels none/high/max, default high, and `can_disable=true` +# (accessed 2026-06-25): https://api.code.umans.ai/v1/models/info reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] [interleaved] diff --git a/providers/umans-ai/models/umans-kimi-k2.7.toml b/providers/umans-ai/models/umans-kimi-k2.7.toml index 02c3a499f..913774482 100644 --- a/providers/umans-ai/models/umans-kimi-k2.7.toml +++ b/providers/umans-ai/models/umans-kimi-k2.7.toml @@ -1,5 +1,7 @@ base_model = "moonshotai/kimi-k2.7-code" name = "Kimi K2.7 Code" +# Live metadata says reasoning is always on (`can_disable=false`) with no effort +# levels (accessed 2026-06-25): https://api.code.umans.ai/v1/models/info reasoning_options = [] [interleaved] diff --git a/providers/umans-ai/provider.toml b/providers/umans-ai/provider.toml index 50c077cd3..d4b28fe25 100644 --- a/providers/umans-ai/provider.toml +++ b/providers/umans-ai/provider.toml @@ -1,5 +1,16 @@ name = "Umans AI" env = ["UMANS_AI_API_KEY"] npm = "@ai-sdk/openai-compatible" +# Reasoning HTTP format (accessed 2026-06-25): +# POST /v1/chat/completions accepts top-level `reasoning_effort`; POST +# /v1/messages accepts `thinking.type` and `thinking.budget_tokens`. Either +# control is accepted on either endpoint; if both are sent, `thinking` wins. +# `thinking.type` is enabled/disabled; effort maps none/low/medium/high, with +# other values such as minimal mapped to the nearest level. A budget must be +# below max output; it is reduced to fit, and too little room disables thinking. +# No fixed numeric budget bounds are published. Chat returns +# `reasoning_content`; Messages returns thinking blocks/thinking_delta events. +# Live model metadata: GET https://api.code.umans.ai/v1/models/info +# Source: https://app.umans.ai/offers/code/docs#reasoning--extended-thinking doc = "https://app.umans.ai/offers/code/docs/orgs" api = "https://api.code.umans.ai/v1" diff --git a/providers/upstage/provider.toml b/providers/upstage/provider.toml index 51b2ffcb7..28bba9113 100644 --- a/providers/upstage/provider.toml +++ b/providers/upstage/provider.toml @@ -1,4 +1,10 @@ name = "Upstage" +# Raw POST /v1/solar/chat/completions uses top-level reasoning_effort. +# solar-pro2: "high" enables reasoning; omission disables it. solar-pro3: +# "low" disables, "medium" (default) budgets 30%, and "high" budgets 60% of +# remaining context, bounded internally as min(max_budget, max(min_budget, +# ratio * remaining_context)); remaining context is max_tokens when supplied. +# https://developers.upstage.ai/docs/capabilities/generate/reasoning (accessed 2026-06-25) env = ["UPSTAGE_API_KEY"] npm = "@ai-sdk/openai-compatible" doc = "https://developers.upstage.ai/docs/apis/chat" diff --git a/providers/v0/provider.toml b/providers/v0/provider.toml index dc37b106a..56adbd061 100644 --- a/providers/v0/provider.toml +++ b/providers/v0/provider.toml @@ -1,4 +1,6 @@ name = "v0" env = ["V0_API_KEY"] npm = "@ai-sdk/vercel" +# These configured IDs use the Model API at https://api.v0.dev/v1 (for example POST /chat/completions); its public contract documents no reasoning toggle, effort, or budget field. https://sdk.vercel.ai/providers/ai-sdk-providers/vercel (accessed 2026-06-25) +# The separate Platform API uses v0-sdk and https://api.v0.app/v2/chats[//messages]; returned message parts may include thinking, but no caller reasoning control is documented. https://v0.dev/docs/api/v2 (accessed 2026-06-25) doc = "https://sdk.vercel.ai/providers/ai-sdk-providers/vercel" diff --git a/providers/venice/models/aion-labs-aion-2-0.toml b/providers/venice/models/aion-labs-aion-2-0.toml index bb1fc1b40..d317752c6 100644 --- a/providers/venice/models/aion-labs-aion-2-0.toml +++ b/providers/venice/models/aion-labs-aion-2-0.toml @@ -6,6 +6,8 @@ reasoning = true tool_call = false open_weights = false +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/arcee-trinity-large-thinking.toml b/providers/venice/models/arcee-trinity-large-thinking.toml index 0b40caf89..b3759953d 100644 --- a/providers/venice/models/arcee-trinity-large-thinking.toml +++ b/providers/venice/models/arcee-trinity-large-thinking.toml @@ -8,6 +8,8 @@ tool_call = true structured_output = true open_weights = true +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/claude-fable-5.toml b/providers/venice/models/claude-fable-5.toml index 67103c85f..d6b2e06e0 100644 --- a/providers/venice/models/claude-fable-5.toml +++ b/providers/venice/models/claude-fable-5.toml @@ -2,6 +2,8 @@ base_model = "anthropic/claude-fable-5" release_date = "2026-06-10" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-5.toml b/providers/venice/models/claude-opus-4-5.toml index 4214b6ea7..036a98944 100644 --- a/providers/venice/models/claude-opus-4-5.toml +++ b/providers/venice/models/claude-opus-4-5.toml @@ -3,6 +3,8 @@ name = "Claude Opus 4.5" release_date = "2025-12-06" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-6-fast.toml b/providers/venice/models/claude-opus-4-6-fast.toml index f6da87c51..01677ed04 100644 --- a/providers/venice/models/claude-opus-4-6-fast.toml +++ b/providers/venice/models/claude-opus-4-6-fast.toml @@ -3,6 +3,8 @@ name = "Claude Opus 4.6 Fast" release_date = "2026-04-08" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-6.toml b/providers/venice/models/claude-opus-4-6.toml index f44bb5903..5439d815d 100644 --- a/providers/venice/models/claude-opus-4-6.toml +++ b/providers/venice/models/claude-opus-4-6.toml @@ -1,6 +1,8 @@ base_model = "anthropic/claude-opus-4-6" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-7-fast.toml b/providers/venice/models/claude-opus-4-7-fast.toml index d3ec897a5..94d4aa219 100644 --- a/providers/venice/models/claude-opus-4-7-fast.toml +++ b/providers/venice/models/claude-opus-4-7-fast.toml @@ -3,6 +3,8 @@ name = "Claude Opus 4.7 Fast" release_date = "2026-05-14" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-7.toml b/providers/venice/models/claude-opus-4-7.toml index b1647e8d5..c1c0fbf80 100644 --- a/providers/venice/models/claude-opus-4-7.toml +++ b/providers/venice/models/claude-opus-4-7.toml @@ -1,6 +1,8 @@ base_model = "anthropic/claude-opus-4-7" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-8-fast.toml b/providers/venice/models/claude-opus-4-8-fast.toml index 800effef9..d9403cf4a 100644 --- a/providers/venice/models/claude-opus-4-8-fast.toml +++ b/providers/venice/models/claude-opus-4-8-fast.toml @@ -2,6 +2,8 @@ base_model = "anthropic/claude-opus-4-8" name = "Claude Opus 4.8 Fast" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-opus-4-8.toml b/providers/venice/models/claude-opus-4-8.toml index f21a63789..bc08ff4f2 100644 --- a/providers/venice/models/claude-opus-4-8.toml +++ b/providers/venice/models/claude-opus-4-8.toml @@ -1,6 +1,8 @@ base_model = "anthropic/claude-opus-4-8" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-sonnet-4-5.toml b/providers/venice/models/claude-sonnet-4-5.toml index 56abf44b9..fe2567eb3 100644 --- a/providers/venice/models/claude-sonnet-4-5.toml +++ b/providers/venice/models/claude-sonnet-4-5.toml @@ -3,6 +3,8 @@ name = "Claude Sonnet 4.5" release_date = "2025-01-15" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/claude-sonnet-4-6.toml b/providers/venice/models/claude-sonnet-4-6.toml index d1ff07091..1cf39500e 100644 --- a/providers/venice/models/claude-sonnet-4-6.toml +++ b/providers/venice/models/claude-sonnet-4-6.toml @@ -1,6 +1,8 @@ base_model = "anthropic/claude-sonnet-4-6" last_updated = "2026-06-11" structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/deepseek-v3.2.toml b/providers/venice/models/deepseek-v3.2.toml index 3154ddc0a..dce9df4e9 100644 --- a/providers/venice/models/deepseek-v3.2.toml +++ b/providers/venice/models/deepseek-v3.2.toml @@ -8,6 +8,8 @@ tool_call = true structured_output = true open_weights = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/deepseek-v4-flash.toml b/providers/venice/models/deepseek-v4-flash.toml index e2387b66a..1b39b2db0 100644 --- a/providers/venice/models/deepseek-v4-flash.toml +++ b/providers/venice/models/deepseek-v4-flash.toml @@ -1,6 +1,8 @@ base_model = "deepseek/deepseek-v4-flash" family = "deepseek" last_updated = "2026-06-11" +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [interleaved] diff --git a/providers/venice/models/deepseek-v4-pro.toml b/providers/venice/models/deepseek-v4-pro.toml index dae00a559..bf79676c8 100644 --- a/providers/venice/models/deepseek-v4-pro.toml +++ b/providers/venice/models/deepseek-v4-pro.toml @@ -1,6 +1,8 @@ base_model = "deepseek/deepseek-v4-pro" family = "deepseek" last_updated = "2026-06-11" +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [interleaved] diff --git a/providers/venice/models/gemini-3-1-pro-preview.toml b/providers/venice/models/gemini-3-1-pro-preview.toml index 3b5e8c760..1e1cf2e00 100644 --- a/providers/venice/models/gemini-3-1-pro-preview.toml +++ b/providers/venice/models/gemini-3-1-pro-preview.toml @@ -2,6 +2,8 @@ base_model = "google/gemini-3.1-pro-preview" family = "gemini" last_updated = "2026-06-11" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/gemini-3-5-flash.toml b/providers/venice/models/gemini-3-5-flash.toml index d8139f77c..d544a943c 100644 --- a/providers/venice/models/gemini-3-5-flash.toml +++ b/providers/venice/models/gemini-3-5-flash.toml @@ -3,6 +3,8 @@ family = "gemini" release_date = "2026-05-22" last_updated = "2026-06-11" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/gemini-3-flash-preview.toml b/providers/venice/models/gemini-3-flash-preview.toml index 1e5eedd40..53c72143b 100644 --- a/providers/venice/models/gemini-3-flash-preview.toml +++ b/providers/venice/models/gemini-3-flash-preview.toml @@ -3,6 +3,8 @@ family = "gemini" release_date = "2025-12-19" last_updated = "2026-06-11" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/google-gemma-4-26b-a4b-it.toml b/providers/venice/models/google-gemma-4-26b-a4b-it.toml index ea28dba64..d6522c7e5 100644 --- a/providers/venice/models/google-gemma-4-26b-a4b-it.toml +++ b/providers/venice/models/google-gemma-4-26b-a4b-it.toml @@ -2,6 +2,8 @@ base_model = "google/gemma-4-26b-a4b-it" name = "Google Gemma 4 26B A4B Instruct" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/google-gemma-4-31b-it.toml b/providers/venice/models/google-gemma-4-31b-it.toml index ee0e46734..d0568d7fb 100644 --- a/providers/venice/models/google-gemma-4-31b-it.toml +++ b/providers/venice/models/google-gemma-4-31b-it.toml @@ -3,6 +3,8 @@ name = "Google Gemma 4 31B Instruct" release_date = "2026-04-03" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/grok-4-20-multi-agent.toml b/providers/venice/models/grok-4-20-multi-agent.toml index 6a96c8018..7aa2ce82e 100644 --- a/providers/venice/models/grok-4-20-multi-agent.toml +++ b/providers/venice/models/grok-4-20-multi-agent.toml @@ -7,6 +7,8 @@ reasoning = true tool_call = false structured_output = true open_weights = false +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/grok-4-20.toml b/providers/venice/models/grok-4-20.toml index b8f505d5a..255453f28 100644 --- a/providers/venice/models/grok-4-20.toml +++ b/providers/venice/models/grok-4-20.toml @@ -7,6 +7,8 @@ reasoning = true tool_call = true structured_output = true open_weights = false +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/grok-4-3.toml b/providers/venice/models/grok-4-3.toml index 7847883fe..06475951f 100644 --- a/providers/venice/models/grok-4-3.toml +++ b/providers/venice/models/grok-4-3.toml @@ -1,6 +1,8 @@ base_model = "xai/grok-4.3" release_date = "2026-04-18" last_updated = "2026-06-11" +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/grok-build-0-1.toml b/providers/venice/models/grok-build-0-1.toml index 1934415d0..217ab0814 100644 --- a/providers/venice/models/grok-build-0-1.toml +++ b/providers/venice/models/grok-build-0-1.toml @@ -1,6 +1,8 @@ base_model = "xai/grok-build-0.1" release_date = "2026-05-21" last_updated = "2026-06-11" +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/kimi-k2-5.toml b/providers/venice/models/kimi-k2-5.toml index a8f58ec48..e1b63bb3e 100644 --- a/providers/venice/models/kimi-k2-5.toml +++ b/providers/venice/models/kimi-k2-5.toml @@ -5,6 +5,8 @@ attachment = true knowledge = "2024-04" open_weights = false +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/kimi-k2-6.toml b/providers/venice/models/kimi-k2-6.toml index 97d84983a..6eb723d30 100644 --- a/providers/venice/models/kimi-k2-6.toml +++ b/providers/venice/models/kimi-k2-6.toml @@ -1,6 +1,8 @@ base_model = "moonshotai/kimi-k2.6" release_date = "2026-04-20" last_updated = "2026-06-11" +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [interleaved] diff --git a/providers/venice/models/kimi-k2-7-code.toml b/providers/venice/models/kimi-k2-7-code.toml index 638f9b19a..d673bfa25 100644 --- a/providers/venice/models/kimi-k2-7-code.toml +++ b/providers/venice/models/kimi-k2-7-code.toml @@ -2,6 +2,8 @@ base_model = "moonshotai/kimi-k2.7-code" release_date = "2026-06-13" last_updated = "2026-06-16" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/mercury-2.toml b/providers/venice/models/mercury-2.toml index 6853ce6ad..e4485dc93 100644 --- a/providers/venice/models/mercury-2.toml +++ b/providers/venice/models/mercury-2.toml @@ -8,6 +8,8 @@ tool_call = true structured_output = true open_weights = false +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/minimax-m25.toml b/providers/venice/models/minimax-m25.toml index 0f1c7b306..da8027047 100644 --- a/providers/venice/models/minimax-m25.toml +++ b/providers/venice/models/minimax-m25.toml @@ -2,6 +2,8 @@ base_model = "minimax/MiniMax-M2.5" name = "MiniMax M2.5" last_updated = "2026-06-11" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/minimax-m27.toml b/providers/venice/models/minimax-m27.toml index 061f0a920..046f0551d 100644 --- a/providers/venice/models/minimax-m27.toml +++ b/providers/venice/models/minimax-m27.toml @@ -2,6 +2,8 @@ base_model = "minimax/MiniMax-M2.7" name = "MiniMax M2.7" last_updated = "2026-06-11" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/minimax-m3-preview.toml b/providers/venice/models/minimax-m3-preview.toml index cfbc71519..60bb53f11 100644 --- a/providers/venice/models/minimax-m3-preview.toml +++ b/providers/venice/models/minimax-m3-preview.toml @@ -6,6 +6,8 @@ attachment = false reasoning = true tool_call = true open_weights = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/mistral-small-2603.toml b/providers/venice/models/mistral-small-2603.toml index 58a3e7d71..238e8672f 100644 --- a/providers/venice/models/mistral-small-2603.toml +++ b/providers/venice/models/mistral-small-2603.toml @@ -2,6 +2,8 @@ base_model = "mistral/mistral-small-2603" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|high only; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "high"] diff --git a/providers/venice/models/nvidia-nemotron-3-ultra-550b-a55b.toml b/providers/venice/models/nvidia-nemotron-3-ultra-550b-a55b.toml index 32e06cb95..033b8c496 100644 --- a/providers/venice/models/nvidia-nemotron-3-ultra-550b-a55b.toml +++ b/providers/venice/models/nvidia-nemotron-3-ultra-550b-a55b.toml @@ -3,6 +3,8 @@ name = "NVIDIA Nemotron 3 Ultra" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/nvidia-nemotron-cascade-2-30b-a3b.toml b/providers/venice/models/nvidia-nemotron-cascade-2-30b-a3b.toml index ac227bdf2..f71c5107d 100644 --- a/providers/venice/models/nvidia-nemotron-cascade-2-30b-a3b.toml +++ b/providers/venice/models/nvidia-nemotron-cascade-2-30b-a3b.toml @@ -2,6 +2,8 @@ base_model = "nvidia/nemotron-cascade-2-30b-a3b" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/olafangensan-glm-4.7-flash-heretic.toml b/providers/venice/models/olafangensan-glm-4.7-flash-heretic.toml index 9c774ff09..a2ec66add 100644 --- a/providers/venice/models/olafangensan-glm-4.7-flash-heretic.toml +++ b/providers/venice/models/olafangensan-glm-4.7-flash-heretic.toml @@ -8,6 +8,8 @@ tool_call = true structured_output = true open_weights = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-52-codex.toml b/providers/venice/models/openai-gpt-52-codex.toml index b19daba97..05e14ddbb 100644 --- a/providers/venice/models/openai-gpt-52-codex.toml +++ b/providers/venice/models/openai-gpt-52-codex.toml @@ -4,6 +4,8 @@ release_date = "2025-01-15" last_updated = "2026-06-11" knowledge = "2025-08" +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-52.toml b/providers/venice/models/openai-gpt-52.toml index 0ccced99d..4f1ada52e 100644 --- a/providers/venice/models/openai-gpt-52.toml +++ b/providers/venice/models/openai-gpt-52.toml @@ -3,6 +3,8 @@ release_date = "2025-12-13" last_updated = "2026-06-11" attachment = false +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-53-codex.toml b/providers/venice/models/openai-gpt-53-codex.toml index db55e2f61..e563f7ebe 100644 --- a/providers/venice/models/openai-gpt-53-codex.toml +++ b/providers/venice/models/openai-gpt-53-codex.toml @@ -3,6 +3,8 @@ family = "gpt" release_date = "2026-02-24" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-54-mini.toml b/providers/venice/models/openai-gpt-54-mini.toml index d4f5e975b..e38e2b467 100644 --- a/providers/venice/models/openai-gpt-54-mini.toml +++ b/providers/venice/models/openai-gpt-54-mini.toml @@ -4,6 +4,8 @@ family = "gpt" release_date = "2026-03-27" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-54-pro.toml b/providers/venice/models/openai-gpt-54-pro.toml index 2254bd9c2..d0a44d756 100644 --- a/providers/venice/models/openai-gpt-54-pro.toml +++ b/providers/venice/models/openai-gpt-54-pro.toml @@ -3,6 +3,8 @@ family = "gpt" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-54.toml b/providers/venice/models/openai-gpt-54.toml index af80be27d..450eaf99b 100644 --- a/providers/venice/models/openai-gpt-54.toml +++ b/providers/venice/models/openai-gpt-54.toml @@ -1,6 +1,8 @@ base_model = "openai/gpt-5.4" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-55-pro.toml b/providers/venice/models/openai-gpt-55-pro.toml index d84cd0bcd..35372f883 100644 --- a/providers/venice/models/openai-gpt-55-pro.toml +++ b/providers/venice/models/openai-gpt-55-pro.toml @@ -3,6 +3,8 @@ family = "gpt" release_date = "2026-04-24" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-55.toml b/providers/venice/models/openai-gpt-55.toml index 83dddec05..931c8bf18 100644 --- a/providers/venice/models/openai-gpt-55.toml +++ b/providers/venice/models/openai-gpt-55.toml @@ -1,6 +1,8 @@ base_model = "openai/gpt-5.5" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|minimal|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "minimal", "low", "medium", "high"] diff --git a/providers/venice/models/openai-gpt-oss-120b.toml b/providers/venice/models/openai-gpt-oss-120b.toml index 0585815e6..2119dd18f 100644 --- a/providers/venice/models/openai-gpt-oss-120b.toml +++ b/providers/venice/models/openai-gpt-oss-120b.toml @@ -7,6 +7,8 @@ reasoning = true tool_call = true open_weights = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/qwen-3-6-plus.toml b/providers/venice/models/qwen-3-6-plus.toml index 96de594ad..cff0dc52e 100644 --- a/providers/venice/models/qwen-3-6-plus.toml +++ b/providers/venice/models/qwen-3-6-plus.toml @@ -4,6 +4,8 @@ release_date = "2026-04-06" last_updated = "2026-06-11" attachment = true structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/qwen-3-7-max.toml b/providers/venice/models/qwen-3-7-max.toml index f971d5ba8..aa1ee9db3 100644 --- a/providers/venice/models/qwen-3-7-max.toml +++ b/providers/venice/models/qwen-3-7-max.toml @@ -3,6 +3,8 @@ name = "Qwen 3.7 Max" release_date = "2026-05-22" last_updated = "2026-06-11" attachment = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/qwen-3-7-plus.toml b/providers/venice/models/qwen-3-7-plus.toml index 6d184e71d..b0d380541 100644 --- a/providers/venice/models/qwen-3-7-plus.toml +++ b/providers/venice/models/qwen-3-7-plus.toml @@ -3,6 +3,8 @@ name = "Qwen 3.7 Plus" last_updated = "2026-06-11" attachment = true structured_output = true +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/qwen3-235b-a22b-thinking-2507.toml b/providers/venice/models/qwen3-235b-a22b-thinking-2507.toml index bf0661f0a..1edb8b7b0 100644 --- a/providers/venice/models/qwen3-235b-a22b-thinking-2507.toml +++ b/providers/venice/models/qwen3-235b-a22b-thinking-2507.toml @@ -8,6 +8,8 @@ tool_call = true structured_output = true open_weights = true +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/qwen3-5-35b-a3b.toml b/providers/venice/models/qwen3-5-35b-a3b.toml index 978faa836..d6ac2b664 100644 --- a/providers/venice/models/qwen3-5-35b-a3b.toml +++ b/providers/venice/models/qwen3-5-35b-a3b.toml @@ -3,6 +3,8 @@ name = "Qwen 3.5 35B A3B" release_date = "2026-02-25" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/qwen3-5-397b-a17b.toml b/providers/venice/models/qwen3-5-397b-a17b.toml index 2a6f2d292..2bf52a57c 100644 --- a/providers/venice/models/qwen3-5-397b-a17b.toml +++ b/providers/venice/models/qwen3-5-397b-a17b.toml @@ -3,6 +3,8 @@ name = "Qwen 3.5 397B" release_date = "2026-02-16" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/qwen3-5-9b.toml b/providers/venice/models/qwen3-5-9b.toml index 68565dc3e..cf33f72fc 100644 --- a/providers/venice/models/qwen3-5-9b.toml +++ b/providers/venice/models/qwen3-5-9b.toml @@ -4,6 +4,8 @@ release_date = "2026-03-05" last_updated = "2026-06-11" attachment = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/qwen3-6-27b.toml b/providers/venice/models/qwen3-6-27b.toml index 7279dce8e..6928e1ff9 100644 --- a/providers/venice/models/qwen3-6-27b.toml +++ b/providers/venice/models/qwen3-6-27b.toml @@ -4,6 +4,8 @@ release_date = "2026-04-24" last_updated = "2026-06-11" open_weights = false +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/xiaomi-mimo-v2-5.toml b/providers/venice/models/xiaomi-mimo-v2-5.toml index 6ceeda83b..27795e6ed 100644 --- a/providers/venice/models/xiaomi-mimo-v2-5.toml +++ b/providers/venice/models/xiaomi-mimo-v2-5.toml @@ -3,6 +3,8 @@ release_date = "2026-06-11" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/z-ai-glm-5-turbo.toml b/providers/venice/models/z-ai-glm-5-turbo.toml index 9e188f51c..616ca56a4 100644 --- a/providers/venice/models/z-ai-glm-5-turbo.toml +++ b/providers/venice/models/z-ai-glm-5-turbo.toml @@ -4,6 +4,8 @@ release_date = "2026-03-15" last_updated = "2026-06-11" open_weights = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/z-ai-glm-5v-turbo.toml b/providers/venice/models/z-ai-glm-5v-turbo.toml index 5b8df4237..f0f72af28 100644 --- a/providers/venice/models/z-ai-glm-5v-turbo.toml +++ b/providers/venice/models/z-ai-glm-5v-turbo.toml @@ -3,6 +3,8 @@ name = "GLM 5V Turbo" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/zai-org-glm-4.6.toml b/providers/venice/models/zai-org-glm-4.6.toml index 76afff384..5aa607388 100644 --- a/providers/venice/models/zai-org-glm-4.6.toml +++ b/providers/venice/models/zai-org-glm-4.6.toml @@ -7,6 +7,8 @@ structured_output = true [interleaved] field = "reasoning_content" +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/zai-org-glm-4.7-flash.toml b/providers/venice/models/zai-org-glm-4.7-flash.toml index a2014abec..9673ef9a9 100644 --- a/providers/venice/models/zai-org-glm-4.7-flash.toml +++ b/providers/venice/models/zai-org-glm-4.7-flash.toml @@ -5,6 +5,8 @@ release_date = "2026-01-29" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/zai-org-glm-4.7.toml b/providers/venice/models/zai-org-glm-4.7.toml index 2f6cd98c4..e1ff44f75 100644 --- a/providers/venice/models/zai-org-glm-4.7.toml +++ b/providers/venice/models/zai-org-glm-4.7.toml @@ -4,6 +4,8 @@ release_date = "2025-12-24" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] diff --git a/providers/venice/models/zai-org-glm-5-1.toml b/providers/venice/models/zai-org-glm-5-1.toml index 48322ee4e..e7240d3d4 100644 --- a/providers/venice/models/zai-org-glm-5-1.toml +++ b/providers/venice/models/zai-org-glm-5-1.toml @@ -2,6 +2,8 @@ base_model = "zhipuai/glm-5.1" name = "GLM 5.1" last_updated = "2026-06-11" +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/models/zai-org-glm-5-2.toml b/providers/venice/models/zai-org-glm-5-2.toml index ff8df8f12..67f5bfbce 100644 --- a/providers/venice/models/zai-org-glm-5-2.toml +++ b/providers/venice/models/zai-org-glm-5-2.toml @@ -2,6 +2,8 @@ base_model = "zhipuai/glm-5.2" name = "GLM 5.2" release_date = "2026-06-16" last_updated = "2026-06-16" +# Live metadata: reasoning only; no effort control or dedicated budget (2026-06-25). +# https://api.venice.ai/api/v1/models?type=text reasoning_options = [] [cost] diff --git a/providers/venice/models/zai-org-glm-5.toml b/providers/venice/models/zai-org-glm-5.toml index ffba2a5a4..cb73140f3 100644 --- a/providers/venice/models/zai-org-glm-5.toml +++ b/providers/venice/models/zai-org-glm-5.toml @@ -4,6 +4,8 @@ release_date = "2026-02-11" last_updated = "2026-06-11" structured_output = true +# Live /models (2026-06-25): none|low|medium|high; no dedicated reasoning budget. +# https://api.venice.ai/api/v1/models?type=text (accessed 2026-06-25) [[reasoning_options]] type = "effort" values = ["none", "low", "medium", "high"] diff --git a/providers/venice/provider.toml b/providers/venice/provider.toml index ee9d0c0d7..31a4291f9 100644 --- a/providers/venice/provider.toml +++ b/providers/venice/provider.toml @@ -1,5 +1,22 @@ name = "Venice AI" env = ["VENICE_API_KEY"] npm = "venice-ai-sdk-provider" +# Raw HTTP reasoning controls (sources accessed 2026-06-25): +# - POST /api/v1/chat/completions accepts either +# `reasoning = { effort = "..." }` or top-level `reasoning_effort`; the flat +# field wins if both are sent. `reasoning = { enabled = false }` is Venice's +# preferred toggle, while supported legacy thinking models also accept +# `venice_parameters = { disable_thinking = true }`. +# - Alpha POST /api/v1/responses is stateless and accepts only nested +# `reasoning = { effort = "..." }`; E2EE models must use Chat instead. +# https://docs.venice.ai/guides/features/reasoning-models +# https://docs.venice.ai/swagger.yaml +# Model-specific effort enums come from live GET /api/v1/models fields +# `supportsReasoningEffort` and `reasoningEffortOptions`. Venice exposes no +# dedicated reasoning-token budget: `max_completion_tokens` caps reasoning plus +# visible output, and `model_spec.maxCompletionTokens` supplies that total cap. +# https://api.venice.ai/api/v1/models?type=text +# https://docs.venice.ai/api-reference/endpoint/models/list +# https://docs.venice.ai/api-reference/endpoint/chat/completions doc = "https://docs.venice.ai" # api = "https://api.venice.ai/api/v1" diff --git a/providers/vercel/models/alibaba/qwen3.7-plus.toml b/providers/vercel/models/alibaba/qwen3.7-plus.toml index 6596d0125..b37e9169f 100644 --- a/providers/vercel/models/alibaba/qwen3.7-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.7-plus.toml @@ -1,3 +1,12 @@ +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://ai-gateway.vercel.sh/v1/chat/completions +# JSON reasoning.enabled: true | false; reasoning.max_tokens: number. +# Routes are alibaba, fireworks, and togetherai; pin with +# providerOptions.gateway.only. The documented pages do not verify the model's +# 1..262144 reasoning.max_tokens bounds or equal behavior on every route. +# Sources: +# https://vercel.com/docs/ai-gateway/sdks-and-apis/openai-chat-completions/advanced +# https://vercel.com/ai-gateway/models/qwen3.7-plus base_model = "alibaba/qwen3.7-plus" name = "Qwen 3.7 Plus" family = "qwen3.7-plus" diff --git a/providers/vercel/provider.toml b/providers/vercel/provider.toml index fb8daf980..f90521733 100644 --- a/providers/vercel/provider.toml +++ b/providers/vercel/provider.toml @@ -1,4 +1,14 @@ name = "Vercel AI Gateway" env = ["AI_GATEWAY_API_KEY"] npm = "@ai-sdk/gateway" +# Reasoning HTTP format (accessed 2026-06-25): +# POST https://ai-gateway.vercel.sh/v1/chat/completions +# JSON reasoning.enabled: true | false; reasoning.effort: "none" | "minimal" | +# "low" | "medium" | "high" | "xhigh"; reasoning.max_tokens: number. +# reasoning.effort and reasoning.max_tokens are mutually exclusive. This is the +# gateway vocabulary, not proof that every model or route implements each value. +# Pin a route with providerOptions.gateway.only: [""] (or order). +# Sources: +# https://vercel.com/docs/ai-gateway/sdks-and-apis/openai-chat-completions/advanced +# https://vercel.com/docs/ai-gateway/models-and-providers/provider-filtering-and-ordering doc = "https://github.com/vercel/ai/tree/5eb85cc45a259553501f535b8ac79a77d0e79223/packages/gateway" diff --git a/providers/vivgrid/models/deepseek-v3.2.toml b/providers/vivgrid/models/deepseek-v3.2.toml index bfacdb88b..095f193b9 100644 --- a/providers/vivgrid/models/deepseek-v3.2.toml +++ b/providers/vivgrid/models/deepseek-v3.2.toml @@ -1,4 +1,8 @@ name = "DeepSeek-V3.2" +# Raw POST /v1/chat/completions is documented for this model, but no reasoning +# request field is specified; /v1/responses limits reasoning to GPT-5/o-series. +# https://docs.vivgrid.com/models/deepseek-v3.2 (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "deepseek" release_date = "2025-12-01" last_updated = "2025-12-01" diff --git a/providers/vivgrid/models/deepseek-v4-pro.toml b/providers/vivgrid/models/deepseek-v4-pro.toml index c05dbb0b3..8f554c931 100644 --- a/providers/vivgrid/models/deepseek-v4-pro.toml +++ b/providers/vivgrid/models/deepseek-v4-pro.toml @@ -1,3 +1,7 @@ +# Raw POST /v1/chat/completions is documented for this model, but no reasoning +# request field is specified; /v1/responses limits reasoning to GPT-5/o-series. +# https://docs.vivgrid.com/models/deepseek-v4-pro (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) base_model = "deepseek/deepseek-v4-pro" [interleaved] diff --git a/providers/vivgrid/models/gemini-3.1-flash-lite-preview.toml b/providers/vivgrid/models/gemini-3.1-flash-lite-preview.toml index 166ccfb92..6c9e059cf 100644 --- a/providers/vivgrid/models/gemini-3.1-flash-lite-preview.toml +++ b/providers/vivgrid/models/gemini-3.1-flash-lite-preview.toml @@ -1,4 +1,8 @@ name = "Gemini 3.1 Flash Lite Preview" +# Raw POST /v1/chat/completions is documented for this model, but no reasoning +# request field is specified; /v1/responses limits reasoning to GPT-5/o-series. +# https://docs.vivgrid.com/models/gemini-3.1-flash-lite-preview (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "gemini-flash-lite" release_date = "2026-03-03" last_updated = "2026-03-03" diff --git a/providers/vivgrid/models/gemini-3.1-pro-preview.toml b/providers/vivgrid/models/gemini-3.1-pro-preview.toml index 0f078ba9b..bbd322014 100644 --- a/providers/vivgrid/models/gemini-3.1-pro-preview.toml +++ b/providers/vivgrid/models/gemini-3.1-pro-preview.toml @@ -1,4 +1,8 @@ name = "Gemini 3.1 Pro Preview" +# Raw POST /v1/chat/completions is documented for this model, but no reasoning +# request field is specified; /v1/responses limits reasoning to GPT-5/o-series. +# https://docs.vivgrid.com/models/gemini-3.1-pro-preview (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "gemini-pro" release_date = "2026-02-19" last_updated = "2026-02-19" diff --git a/providers/vivgrid/models/gpt-5-mini.toml b/providers/vivgrid/models/gpt-5-mini.toml index ea9bf5bfd..2009c518e 100644 --- a/providers/vivgrid/models/gpt-5-mini.toml +++ b/providers/vivgrid/models/gpt-5-mini.toml @@ -1,4 +1,9 @@ name = "GPT-5 Mini" +# Raw POST /v1/chat/completions is documented for this ID without a reasoning +# field. POST /v1/responses has an opaque reasoning object for GPT-5/o-series, +# but its documented model list excludes this ID, so no control is verified. +# https://docs.vivgrid.com/models/gpt-5-mini (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "gpt-mini" release_date = "2025-08-07" last_updated = "2025-08-07" diff --git a/providers/vivgrid/models/gpt-5.4-mini.toml b/providers/vivgrid/models/gpt-5.4-mini.toml index da8fda60c..2044fff50 100644 --- a/providers/vivgrid/models/gpt-5.4-mini.toml +++ b/providers/vivgrid/models/gpt-5.4-mini.toml @@ -1,4 +1,9 @@ name = "GPT-5.4 Mini" +# Raw POST /v1/chat/completions is documented for this ID without a reasoning +# field. POST /v1/responses has an opaque reasoning object for GPT-5/o-series, +# but its documented model list excludes this ID, so no control is verified. +# https://docs.vivgrid.com/models/gpt-5.4-mini (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "gpt-mini" release_date = "2026-03-17" last_updated = "2026-03-17" diff --git a/providers/vivgrid/models/gpt-5.4-nano.toml b/providers/vivgrid/models/gpt-5.4-nano.toml index 96d8859dc..f1aaf9b3e 100644 --- a/providers/vivgrid/models/gpt-5.4-nano.toml +++ b/providers/vivgrid/models/gpt-5.4-nano.toml @@ -1,4 +1,9 @@ name = "GPT-5.4 Nano" +# Raw POST /v1/chat/completions is documented for this ID without a reasoning +# field. POST /v1/responses has an opaque reasoning object for GPT-5/o-series, +# but its documented model list excludes this ID, so no control is verified. +# https://docs.vivgrid.com/models/gpt-5.4-nano (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "gpt-nano" release_date = "2026-03-17" last_updated = "2026-03-17" diff --git a/providers/vivgrid/models/gpt-5.4.toml b/providers/vivgrid/models/gpt-5.4.toml index 359e51733..c62a1d291 100644 --- a/providers/vivgrid/models/gpt-5.4.toml +++ b/providers/vivgrid/models/gpt-5.4.toml @@ -1,4 +1,9 @@ name = "GPT-5.4" +# Raw POST /v1/chat/completions is documented for this ID without a reasoning +# field. POST /v1/responses has an opaque reasoning object for GPT-5/o-series, +# but its documented model list excludes this ID, so no control is verified. +# https://docs.vivgrid.com/models/gpt-5.4 (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) family = "gpt" release_date = "2026-03-05" last_updated = "2026-03-05" diff --git a/providers/vivgrid/models/gpt-5.5.toml b/providers/vivgrid/models/gpt-5.5.toml index cf0b548a5..537b352fb 100644 --- a/providers/vivgrid/models/gpt-5.5.toml +++ b/providers/vivgrid/models/gpt-5.5.toml @@ -1,3 +1,8 @@ +# Raw POST /v1/chat/completions is documented for this ID without a reasoning +# field. POST /v1/responses has an opaque reasoning object for GPT-5/o-series, +# but its documented model list excludes this ID, so no control is verified. +# https://docs.vivgrid.com/models/gpt-5.5 (accessed 2026-06-25) +# https://docs.vivgrid.com/api/model-api (accessed 2026-06-25) base_model = "openai/gpt-5.5" [cost] diff --git a/providers/vultr/provider.toml b/providers/vultr/provider.toml index af9a2fb92..c8110a03a 100644 --- a/providers/vultr/provider.toml +++ b/providers/vultr/provider.toml @@ -1,4 +1,8 @@ name = "Vultr" +# POST /v1/chat/completions documents no reasoning request field or control. +# Responses may contain choices[].message.reasoning; a -normalize model suffix +# maps upstream reasoning_content to reasoning but does not control reasoning. +# https://api.vultrinference.com/ (accessed 2026-06-25) npm = "@ai-sdk/openai-compatible" api = "https://api.vultrinference.com/v1" env = ["VULTR_API_KEY"] diff --git a/providers/wafer.ai/provider.toml b/providers/wafer.ai/provider.toml index 52f96f625..4acd6c8ce 100644 --- a/providers/wafer.ai/provider.toml +++ b/providers/wafer.ai/provider.toml @@ -1,4 +1,9 @@ name = "Wafer" +# POST /v1/chat/completions normalizes thinking.type="enabled"|"disabled", +# reasoning_effort="none"|"low"|"medium"|"high"|"max", and +# enable_thinking=true|false. Reasoning defaults off; output is generally +# choices[].message.reasoning_content. Kimi-K2.7-Code is always on. +# https://docs.wafer.ai/serverless/api-reference (accessed 2026-06-25) env = ["WAFER_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://pass.wafer.ai/v1" diff --git a/providers/wandb/provider.toml b/providers/wandb/provider.toml index 9da697afa..a1b023378 100644 --- a/providers/wandb/provider.toml +++ b/providers/wandb/provider.toml @@ -1,4 +1,10 @@ name = "Weights & Biases" +# POST /v1/chat/completions toggles eligible models with +# chat_template_kwargs.enable_thinking=true|false; output is choices[].message.reasoning. +# DeepSeek V4 and Gemma 4 default off; Nemotron 3, Qwen3.5/3.6, and GLM-5.1/5.2 +# default on. MiniMax M2.5, Kimi K2.5/K2.6/K2.7-Code, GPT-OSS, and Qwen3 +# Thinking are always on and cannot be disabled. +# https://docs.wandb.ai/inference/response-settings/reasoning (accessed 2026-06-25) env = ["WANDB_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.inference.wandb.ai/v1" diff --git a/providers/xiaomi-token-plan-ams/provider.toml b/providers/xiaomi-token-plan-ams/provider.toml index e9b087d89..76d5198e0 100644 --- a/providers/xiaomi-token-plan-ams/provider.toml +++ b/providers/xiaomi-token-plan-ams/provider.toml @@ -1,4 +1,10 @@ name = "Xiaomi Token Plan (Europe)" +# Raw endpoints are POST https://token-plan-ams.xiaomimimo.com/v1/chat/completions +# and POST https://token-plan-ams.xiaomimimo.com/anthropic/v1/messages. +# Both protocols use thinking.type="enabled"|"disabled"; Flash defaults +# disabled, while V2.5 Pro, V2.5, V2 Pro, and V2 Omni default enabled. +# https://platform.xiaomimimo.com/static/docs/price/tokenplan/quick-access.md (accessed 2026-06-25) +# https://platform.xiaomimimo.com/static/docs/api/chat/openai-api.md (accessed 2026-06-25) env = ["XIAOMI_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://token-plan-ams.xiaomimimo.com/v1" diff --git a/providers/xiaomi-token-plan-cn/provider.toml b/providers/xiaomi-token-plan-cn/provider.toml index a89f3b9d4..db1a2b484 100644 --- a/providers/xiaomi-token-plan-cn/provider.toml +++ b/providers/xiaomi-token-plan-cn/provider.toml @@ -1,4 +1,10 @@ name = "Xiaomi Token Plan (China)" +# Raw endpoints are POST https://token-plan-cn.xiaomimimo.com/v1/chat/completions +# and POST https://token-plan-cn.xiaomimimo.com/anthropic/v1/messages. +# Both protocols use thinking.type="enabled"|"disabled"; Flash defaults +# disabled, while V2.5 Pro, V2.5, V2 Pro, and V2 Omni default enabled. +# https://platform.xiaomimimo.com/static/docs/price/tokenplan/quick-access.md (accessed 2026-06-25) +# https://platform.xiaomimimo.com/static/docs/api/chat/openai-api.md (accessed 2026-06-25) env = ["XIAOMI_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://token-plan-cn.xiaomimimo.com/v1" diff --git a/providers/xiaomi-token-plan-sgp/provider.toml b/providers/xiaomi-token-plan-sgp/provider.toml index 2317751bb..bfffbfed2 100644 --- a/providers/xiaomi-token-plan-sgp/provider.toml +++ b/providers/xiaomi-token-plan-sgp/provider.toml @@ -1,4 +1,10 @@ name = "Xiaomi Token Plan (Singapore)" +# Raw endpoints are POST https://token-plan-sgp.xiaomimimo.com/v1/chat/completions +# and POST https://token-plan-sgp.xiaomimimo.com/anthropic/v1/messages. +# Both protocols use thinking.type="enabled"|"disabled"; Flash defaults +# disabled, while V2.5 Pro, V2.5, V2 Pro, and V2 Omni default enabled. +# https://platform.xiaomimimo.com/static/docs/price/tokenplan/quick-access.md (accessed 2026-06-25) +# https://platform.xiaomimimo.com/static/docs/api/chat/openai-api.md (accessed 2026-06-25) env = ["XIAOMI_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://token-plan-sgp.xiaomimimo.com/v1" diff --git a/providers/xiaomi/provider.toml b/providers/xiaomi/provider.toml index 853161e77..9630b7517 100644 --- a/providers/xiaomi/provider.toml +++ b/providers/xiaomi/provider.toml @@ -1,4 +1,10 @@ name = "Xiaomi" +# Raw endpoints are POST https://api.xiaomimimo.com/v1/chat/completions and +# POST https://api.xiaomimimo.com/anthropic/v1/messages. Both use +# thinking.type="enabled"|"disabled". Flash defaults disabled; V2.5 Pro, +# V2.5, V2 Pro, and V2 Omni default enabled; TTS models do not support it. +# https://platform.xiaomimimo.com/static/docs/api/chat/openai-api.md (accessed 2026-06-25) +# https://platform.xiaomimimo.com/static/docs/api/chat/anthropic-api.md (accessed 2026-06-25) env = ["XIAOMI_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.xiaomimimo.com/v1" diff --git a/providers/xpersona/provider.toml b/providers/xpersona/provider.toml index 6b042c82c..0f35861d7 100644 --- a/providers/xpersona/provider.toml +++ b/providers/xpersona/provider.toml @@ -1,4 +1,7 @@ name = "Xpersona" +# POST /v1/chat/completions accepts reasoning.effort set to "low", "medium", +# "high", "xhigh", or "max"; no disable sentinel or token budget is documented. +# https://www.xpersona.co/api/v1/openapi/ai-public (accessed 2026-06-25) env = ["XPERSONA_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://www.xpersona.co/v1" diff --git a/providers/zai-coding-plan/provider.toml b/providers/zai-coding-plan/provider.toml index 591478fc9..c4351ce99 100644 --- a/providers/zai-coding-plan/provider.toml +++ b/providers/zai-coding-plan/provider.toml @@ -1,3 +1,6 @@ +# OpenAI Chat endpoint: POST https://api.z.ai/api/coding/paas/v4/chat/completions. +# GLM-5.2 maps low|medium|high to high and xhigh|max|ultracode to max. +# https://docs.z.ai/devpack/latest-model (accessed 2026-06-25) name = "Z.AI Coding Plan" env = ["ZHIPU_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/zai/models/glm-5.2.toml b/providers/zai/models/glm-5.2.toml index 7a95acf28..b16101985 100644 --- a/providers/zai/models/glm-5.2.toml +++ b/providers/zai/models/glm-5.2.toml @@ -1,4 +1,7 @@ base_model = "zhipuai/glm-5.2" +# `reasoning_effort`: none|minimal skip thinking, low|medium map to high, +# and xhigh maps to max; effective levels are high|max (default: max). +# https://docs.z.ai/api-reference/llm/chat-completion (accessed 2026-06-25) [[reasoning_options]] type = "effort" diff --git a/providers/zai/provider.toml b/providers/zai/provider.toml index 70d629506..5722a3013 100644 --- a/providers/zai/provider.toml +++ b/providers/zai/provider.toml @@ -1,3 +1,6 @@ +# Chat reasoning: POST https://api.z.ai/api/paas/v4/chat/completions with +# `thinking.type = "enabled" | "disabled"` (default: enabled). +# https://docs.z.ai/guides/capabilities/thinking (accessed 2026-06-25) name = "Z.AI" env = ["ZHIPU_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/zeldoc/models/z-code.toml b/providers/zeldoc/models/z-code.toml index 711964fbb..517f03b56 100644 --- a/providers/zeldoc/models/z-code.toml +++ b/providers/zeldoc/models/z-code.toml @@ -1,3 +1,5 @@ +# Zeldoc publishes no API reference or documented toggle, effort, or token budget. +# https://docs.zeldoc.ai/ (accessed 2026-06-25) name = "Z-Code" release_date = "2026-04-15" last_updated = "2026-04-15" diff --git a/providers/zenmux/models/google/gemini-2.5-flash-lite.toml b/providers/zenmux/models/google/gemini-2.5-flash-lite.toml index 836589162..8f1440799 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash-lite.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash-lite.toml @@ -1,3 +1,5 @@ +# Vertex `thinkingBudget`: -1 (dynamic) or 512..24576; no disable sentinel is listed. +# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) name = "Gemini 2.5 Flash Lite" release_date = "2025-07-22" last_updated = "2025-07-22" diff --git a/providers/zenmux/models/google/gemini-2.5-flash.toml b/providers/zenmux/models/google/gemini-2.5-flash.toml index 41ee3cba2..13c3c66d5 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash.toml @@ -1,3 +1,5 @@ +# Vertex `thinkingBudget`: -1 (dynamic) or 0..24576; 0 disables thinking. +# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) name = "Gemini 2.5 Flash" release_date = "2025-06-17" last_updated = "2025-06-17" diff --git a/providers/zenmux/models/google/gemini-2.5-pro.toml b/providers/zenmux/models/google/gemini-2.5-pro.toml index 48e88f66e..938e7dc2f 100644 --- a/providers/zenmux/models/google/gemini-2.5-pro.toml +++ b/providers/zenmux/models/google/gemini-2.5-pro.toml @@ -1,3 +1,5 @@ +# Vertex `thinkingBudget`: -1 (dynamic) or 128..32768; thinking cannot be disabled. +# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) name = "Gemini 2.5 Pro" release_date = "2025-06-17" last_updated = "2025-06-17" diff --git a/providers/zenmux/provider.toml b/providers/zenmux/provider.toml index e082af1bf..d78496060 100644 --- a/providers/zenmux/provider.toml +++ b/providers/zenmux/provider.toml @@ -1,3 +1,10 @@ +# OpenAI POST /api/v1/chat/completions: `reasoning_effort`, or `reasoning` +# with `effort`, `max_tokens`, and `enabled`. +# Anthropic POST /api/anthropic/v1/messages: `thinking.type` enabled|disabled; +# enabled requires `budget_tokens >= 1024` and `< max_tokens`. +# Vertex POST /api/vertex-ai/v1/publishers/{provider}/models/{model}:generateContent: +# `generationConfig.thinkingConfig` uses `thinkingBudget` or `thinkingLevel`. +# https://docs.zenmux.ai/guide/advanced/reasoning.html (accessed 2026-06-25) name = "ZenMux" env = ["ZENMUX_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/zhipuai-coding-plan/provider.toml b/providers/zhipuai-coding-plan/provider.toml index 041467701..2eac168c6 100644 --- a/providers/zhipuai-coding-plan/provider.toml +++ b/providers/zhipuai-coding-plan/provider.toml @@ -1,3 +1,6 @@ +# OpenAI Chat endpoint: POST https://open.bigmodel.cn/api/coding/paas/v4/chat/completions. +# GLM-5.2 maps low|medium|high to high and xhigh|max|ultracode to max. +# https://docs.bigmodel.cn/cn/coding-plan/latest-model (accessed 2026-06-25) name = "Zhipu AI Coding Plan" env = ["ZHIPU_API_KEY"] npm = "@ai-sdk/openai-compatible" diff --git a/providers/zhipuai/models/glm-5.2.toml b/providers/zhipuai/models/glm-5.2.toml index 6c23838f5..5a97e553f 100644 --- a/providers/zhipuai/models/glm-5.2.toml +++ b/providers/zhipuai/models/glm-5.2.toml @@ -1,3 +1,6 @@ +# `reasoning_effort`: none|minimal skip thinking, low|medium map to high, +# and xhigh maps to max; effective levels are high|max (default: max). +# https://docs.bigmodel.cn/cn/guide/capabilities/thinking (accessed 2026-06-25) name = "GLM-5.2" family = "glm" release_date = "2026-06-13" diff --git a/providers/zhipuai/provider.toml b/providers/zhipuai/provider.toml index 780f0e9c1..3d77b1c52 100644 --- a/providers/zhipuai/provider.toml +++ b/providers/zhipuai/provider.toml @@ -1,3 +1,6 @@ +# Chat reasoning: POST https://open.bigmodel.cn/api/paas/v4/chat/completions with +# `thinking.type = "enabled" | "disabled"` (default: enabled). +# https://docs.bigmodel.cn/cn/guide/capabilities/thinking (accessed 2026-06-25) name = "Zhipu AI" env = ["ZHIPU_API_KEY"] npm = "@ai-sdk/openai-compatible"