Compare commits
245 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 89b834086a | |||
| 1ab2ff8163 | |||
| 9468676683 | |||
| 6cdd2f054b | |||
| b13abc9141 | |||
| e5ba264751 | |||
| a7811fb522 | |||
| 605fae75d9 | |||
| 26b05268ae | |||
| 1aee13d2e5 | |||
| 0a924e6bb2 | |||
| 9769b2b11d | |||
| 1b0db099cf | |||
| 16ed78587c | |||
| 0b88965165 | |||
| acc704ce39 | |||
| 51ad3b264e | |||
| 146b6c7084 | |||
| 0e3cbe3c64 | |||
| 604d4d66a4 | |||
| f5090028b8 | |||
| 4bad8faf29 | |||
| 0df2ccf586 | |||
| bafdc00b45 | |||
| 49840c013b | |||
| eccae0b54e | |||
| 4cca29405f | |||
| e40d9dd338 | |||
| 6a74991397 | |||
| 035999cb58 | |||
| cec56bf1bc | |||
| 85f0cdcb2f | |||
| ef80d4df4e | |||
| af0ef00109 | |||
| 5fdcea6b36 | |||
| 31e56480b4 | |||
| 92a621594e | |||
| d2db353ceb | |||
| 900ae509d2 | |||
| 9d60164243 | |||
| cd3e99025f | |||
| 1098981eb6 | |||
| 27a151cf53 | |||
| 41ff42ab7f | |||
| adf1cbdecd | |||
| 10ddc78ce0 | |||
| 7ae897e440 | |||
| 02e452c2e8 | |||
| f05b63fff5 | |||
| e277d60236 | |||
| b01277a737 | |||
| a5da5aa429 | |||
| 5ceac8a58b | |||
| 11e1d5623a | |||
| f107afc57c | |||
| e789d7c1f2 | |||
| 899668ad49 | |||
| 8c677f0134 | |||
| 462c7877d9 | |||
| 55871f9dca | |||
| 4f7194a3c8 | |||
| f65f0148da | |||
| 14952f8855 | |||
| d7c6d3ad12 | |||
| 356bc79d08 | |||
| a89b1ed726 | |||
| a998576773 | |||
| fbe842bbea | |||
| 6c0c3d1b10 | |||
| db0a7cf611 | |||
| d775e37e3b | |||
| 36753063d9 | |||
| 5ee955297a | |||
| 1b77511903 | |||
| 8896ead7bf | |||
| eb96594d47 | |||
| 327332efe3 | |||
| 5020951745 | |||
| cb6f97774e | |||
| 7f8b493b0c | |||
| d65a862533 | |||
| 9420048dfe | |||
| 8db6c27634 | |||
| 8e710e19ea | |||
| 36c6896e97 | |||
| a8be548a5d | |||
| 8c2fae4ab0 | |||
| 5e7fad350d | |||
| feb85ef2c9 | |||
| d4161ebf24 | |||
| b91ab02e2b | |||
| dd09d07f75 | |||
| 4515f85d47 | |||
| 97572240e1 | |||
| 80cc04e9e5 | |||
| d532ebb89b | |||
| 9bae3e887f | |||
| 0be69bf872 | |||
| 0ed38cbecf | |||
| e65703382a | |||
| 748754c99b | |||
| ec9c12d0fc | |||
| ac81822c89 | |||
| d32ed764bd | |||
| 45fb951c42 | |||
| 260d79b2d5 | |||
| 746b9caf79 | |||
| 8362b55503 | |||
| dde3953a9f | |||
| 0a1695212c | |||
| 89fbb6bb69 | |||
| 005fe0fb5a | |||
| e283875ce7 | |||
| 3598019251 | |||
| e9ad8b0a3f | |||
| 0ee78eeda5 | |||
| 0a5b33e518 | |||
| 22416dda64 | |||
| eba2702e3a | |||
| c2c5cc8f21 | |||
| 022b1b9946 | |||
| 8269e04222 | |||
| a75cf2ed1c | |||
| 7b00aafa79 | |||
| e54e0fc8b5 | |||
| 38611e75fa | |||
| e62c1e973e | |||
| c2b3c601e4 | |||
| 9ff1d36a21 | |||
| e0f4042ad1 | |||
| 14736ba4b6 | |||
| 9351d68731 | |||
| 7a5a1d2aff | |||
| 699284ce91 | |||
| 21ce5c3ac8 | |||
| 3485cf52d0 | |||
| 99e8f25c78 | |||
| 7102978cb4 | |||
| 530f60c69c | |||
| 2415c5be21 | |||
| 85aba468cd | |||
| d1ec1ba777 | |||
| 0f94bf16ec | |||
| 506e8f48a9 | |||
| 3480bc5992 | |||
| fde97814ef | |||
| ff7eddcb70 | |||
| 5cbab85b8d | |||
| 6f9820de9f | |||
| cdfb429098 | |||
| 1c2546af8a | |||
| 30b3e677fd | |||
| 3b37eee86e | |||
| 71f069670e | |||
| f401672689 | |||
| 2a0d86a034 | |||
| d9439cdf2f | |||
| 3d443d568d | |||
| a76c8fe9dd | |||
| d08e8d6cc1 | |||
| 4c06e44047 | |||
| 5e344ded49 | |||
| 458b7f4d1a | |||
| baf4432140 | |||
| bbf72ea4e0 | |||
| 8f9adc7567 | |||
| addaf1c036 | |||
| 55d16a58b6 | |||
| 72a4deab66 | |||
| 8dd829a187 | |||
| a671cc05d5 | |||
| 656c6f08a7 | |||
| 8979741a32 | |||
| 82851b9a3d | |||
| ae511892d7 | |||
| 9312418242 | |||
| 122627a852 | |||
| 6883e793ce | |||
| e69064709b | |||
| bd8e582b96 | |||
| 21945db90f | |||
| 1771e02be8 | |||
| bb08fc26e9 | |||
| 2e015de42d | |||
| 914a3d9d18 | |||
| bab01dd9ab | |||
| bdaae956af | |||
| 6f118145c0 | |||
| 386eaed119 | |||
| 151e9c9071 | |||
| ba99f1edce | |||
| b96b074a3b | |||
| 2593e131a1 | |||
| 4aebbe5ca3 | |||
| b98495e9c8 | |||
| bc95b42ccd | |||
| 6139fb8c69 | |||
| 3070758007 | |||
| 5525e83de4 | |||
| 359fd879b8 | |||
| 01b5a1a656 | |||
| b1958be099 | |||
| 08aa068523 | |||
| f31ad0b02f | |||
| 585aa7fa1b | |||
| c42a327b3e | |||
| 92ebbfb5c4 | |||
| 535fe8c971 | |||
| a3b4bfc16c | |||
| d0fcd6f11f | |||
| e55cd54218 | |||
| 0d73b82b9f | |||
| 83c7e2b63f | |||
| 83ae4cf813 | |||
| 77eae6eef7 | |||
| 8cbf6ed10e | |||
| 06d87e4411 | |||
| 2cb3832618 | |||
| df960d1a90 | |||
| 8f2f83ef61 | |||
| b133426465 | |||
| dafff5a770 | |||
| dd894f077f | |||
| 91590874e7 | |||
| a436236146 | |||
| 1415b4be97 | |||
| 70ac6fccda | |||
| dc3283417d | |||
| 34fd6673e5 | |||
| 175082d43f | |||
| 0f1855c0a7 | |||
| 28c0d9ce23 | |||
| c5fbcc2c9b | |||
| ab2eb51b4e | |||
| 8e19ec580c | |||
| 3aecc94c46 | |||
| 90515f1913 | |||
| 6c3c4a721c | |||
| c59c4aae6c | |||
| 2cb5a99b98 | |||
| 0c2e47e8ba | |||
| dbe92646c3 | |||
| 4717c67054 | |||
| 16a8fa5c20 | |||
| 7336b3619c |
@@ -35,3 +35,4 @@ jobs:
|
||||
- run: bun sst deploy --stage=dev
|
||||
env:
|
||||
CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }}
|
||||
CLOUDFLARE_DEFAULT_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_DEFAULT_ACCOUNT_ID }}
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
name: Sync Model Catalogs
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "17 * * * *"
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency: ${{ github.workflow }}-${{ github.ref }}
|
||||
|
||||
jobs:
|
||||
providers:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
matrix: ${{ steps.providers.outputs.matrix }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
|
||||
with:
|
||||
ref: dev
|
||||
|
||||
- name: Setup Bun
|
||||
uses: oven-sh/setup-bun@f4d14e03ff726c06358e5557344e1da148b56cf7
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install dependencies
|
||||
run: bun install
|
||||
|
||||
- name: List sync providers
|
||||
id: providers
|
||||
run: echo "matrix=$(bun models:sync --list-providers)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
sync:
|
||||
needs: providers
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.providers.outputs.matrix) }}
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5
|
||||
with:
|
||||
ref: dev
|
||||
|
||||
- name: Setup Bun
|
||||
uses: oven-sh/setup-bun@f4d14e03ff726c06358e5557344e1da148b56cf7
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install dependencies
|
||||
run: bun install
|
||||
|
||||
- name: Sync model catalogs
|
||||
run: bun models:sync ${{ matrix.provider }}
|
||||
env:
|
||||
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
|
||||
GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
GOOGLE_GENERATIVE_AI_API_KEY: ${{ secrets.GOOGLE_GENERATIVE_AI_API_KEY }}
|
||||
XAI_API_KEY: ${{ secrets.XAI_API_KEY }}
|
||||
CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID }}
|
||||
CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN: ${{ secrets.CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN }}
|
||||
|
||||
- name: Validate models
|
||||
run: bun validate
|
||||
|
||||
- name: Create pull request
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
BRANCH: automation/sync-models-${{ matrix.provider }}
|
||||
LABELS: automation,model-sync,provider:${{ matrix.provider }}
|
||||
TITLE: "chore(sync): update ${{ matrix.name }} model catalog"
|
||||
run: |
|
||||
if [ -z "$(git status --porcelain -- providers)" ]; then
|
||||
echo "No model catalog changes found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
git checkout -B "$BRANCH"
|
||||
git add providers
|
||||
git commit -m "$TITLE"
|
||||
git push --force-with-lease origin "$BRANCH"
|
||||
|
||||
label_args=()
|
||||
IFS=',' read -ra labels <<< "$LABELS"
|
||||
for label in "${labels[@]}"; do
|
||||
gh label create "$label" --color "0E8A16" --description "Automated model catalog sync" >/dev/null 2>&1 || true
|
||||
label_args+=(--label "$label")
|
||||
done
|
||||
|
||||
pr_number="$(gh pr list --head "$BRANCH" --base dev --json number --jq '.[0].number')"
|
||||
if [ -n "$pr_number" ]; then
|
||||
gh pr edit "$pr_number" --title "$TITLE" --body-file .sync/model-sync-report.md
|
||||
for label in "${labels[@]}"; do
|
||||
gh pr edit "$pr_number" --add-label "$label"
|
||||
done
|
||||
else
|
||||
gh pr create --base dev --head "$BRANCH" --title "$TITLE" --body-file .sync/model-sync-report.md "${label_args[@]}"
|
||||
fi
|
||||
@@ -3,6 +3,7 @@
|
||||
.idea
|
||||
dist
|
||||
.DS_Store
|
||||
.sync/
|
||||
node_modules
|
||||
data/tokenspeed-monitor.sqlite
|
||||
data/tokenspeed-monitor.sqlite-shm
|
||||
|
||||
File diff suppressed because one or more lines are too long
+6
-1
@@ -17,11 +17,16 @@
|
||||
"scripts": {
|
||||
"validate": "bun ./packages/core/script/validate.ts",
|
||||
"compare:migrations": "bun ./packages/core/script/compare-model-migrations.ts",
|
||||
"cloudflare:sync": "bun ./packages/core/script/sync-models.ts cloudflare-workers-ai",
|
||||
"chutes:generate": "bun ./packages/core/script/generate-chutes.ts",
|
||||
"databricks:generate": "bun ./packages/core/script/generate-databricks.ts",
|
||||
"helicone:generate": "bun ./packages/core/script/generate-helicone.ts",
|
||||
"venice:generate": "bun ./packages/core/script/generate-venice.ts",
|
||||
"vercel:generate": "bun ./packages/core/script/generate-vercel.ts",
|
||||
"wandb:generate": "bun ./packages/core/script/generate-wandb.ts",
|
||||
"digitalocean:generate": "bun ./packages/core/script/generate-digitalocean.ts"
|
||||
"digitalocean:generate": "bun ./packages/core/script/generate-digitalocean.ts",
|
||||
"ambient:generate": "bun ./packages/core/script/generate-ambient.ts",
|
||||
"models:sync": "bun ./packages/core/script/sync-models.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
"@cloudflare/workers-types": "^4.20260424.1",
|
||||
|
||||
@@ -0,0 +1,172 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
/**
|
||||
* Generates Ambient model TOML files from https://api.ambient.xyz/v1/models.
|
||||
*
|
||||
* Emits `[extends]`-format TOMLs that inherit upstream metadata
|
||||
* (family, release_date, knowledge, capabilities) from the canonical
|
||||
* provider model, and override only the fields Ambient's API reports:
|
||||
* cost, limit, modalities.
|
||||
*
|
||||
* Flags:
|
||||
* --dry-run Preview generated TOMLs without writing files.
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { mkdir } from "node:fs/promises";
|
||||
|
||||
const API_ENDPOINT = "https://api.ambient.xyz/v1/models";
|
||||
|
||||
// Allowlist for the initial rollout.
|
||||
const ALLOWLIST = new Set<string>([
|
||||
"zai-org/GLM-5.1-FP8",
|
||||
"moonshotai/kimi-k2.6",
|
||||
]);
|
||||
|
||||
// Maps Ambient model IDs to canonical <provider>/<model> in this repo.
|
||||
// The generated TOML uses this path as `[extends].from` so capabilities
|
||||
// and metadata propagate from the upstream provider automatically.
|
||||
const EXTENDS_MAP: Record<string, string> = {
|
||||
"zai-org/GLM-5.1-FP8": "zai/glm-5.1",
|
||||
"moonshotai/kimi-k2.6": "moonshotai/kimi-k2.6",
|
||||
};
|
||||
|
||||
const Pricing = z
|
||||
.object({
|
||||
prompt: z.string(),
|
||||
completion: z.string(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const AmbientModel = z
|
||||
.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
context_length: z.number(),
|
||||
max_output_length: z.number(),
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
pricing: Pricing,
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const AmbientResponse = z
|
||||
.object({
|
||||
object: z.literal("list"),
|
||||
data: z.array(AmbientModel),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const ALLOWED_MODALITIES = new Set(["text", "audio", "image", "video", "pdf"]);
|
||||
|
||||
function modalities(values: string[]): string[] {
|
||||
return values
|
||||
.map((v) => v.toLowerCase())
|
||||
.filter((v) => ALLOWED_MODALITIES.has(v));
|
||||
}
|
||||
|
||||
function perMTok(price: string): number {
|
||||
const n = parseFloat(price);
|
||||
if (!Number.isFinite(n)) {
|
||||
throw new Error(`Invalid price: ${price}`);
|
||||
}
|
||||
// Round to 6 decimals to absorb float noise from per-token strings.
|
||||
return Math.round(n * 1_000_000 * 1_000_000) / 1_000_000;
|
||||
}
|
||||
|
||||
function formatToml(
|
||||
model: z.infer<typeof AmbientModel>,
|
||||
extendsFrom: string,
|
||||
): string {
|
||||
const lines: string[] = [];
|
||||
lines.push("[extends]");
|
||||
lines.push(`from = "${extendsFrom}"`);
|
||||
lines.push("");
|
||||
|
||||
lines.push("[cost]");
|
||||
lines.push(`input = ${perMTok(model.pricing.prompt)}`);
|
||||
lines.push(`output = ${perMTok(model.pricing.completion)}`);
|
||||
if (model.pricing.input_cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${perMTok(model.pricing.input_cache_read)}`);
|
||||
}
|
||||
if (model.pricing.input_cache_write !== undefined) {
|
||||
lines.push(`cache_write = ${perMTok(model.pricing.input_cache_write)}`);
|
||||
}
|
||||
lines.push("");
|
||||
|
||||
lines.push("[limit]");
|
||||
lines.push(`context = ${model.context_length}`);
|
||||
lines.push(`output = ${model.max_output_length}`);
|
||||
lines.push("");
|
||||
|
||||
const input = modalities(model.input_modalities);
|
||||
const output = modalities(model.output_modalities);
|
||||
lines.push("[modalities]");
|
||||
lines.push(`input = [${input.map((m) => `"${m}"`).join(", ")}]`);
|
||||
lines.push(`output = [${output.map((m) => `"${m}"`).join(", ")}]`);
|
||||
|
||||
return lines.join("\n") + "\n";
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const dryRun = process.argv.includes("--dry-run");
|
||||
|
||||
const outDir = path.join(
|
||||
import.meta.dirname,
|
||||
"..",
|
||||
"..",
|
||||
"..",
|
||||
"providers",
|
||||
"ambient",
|
||||
"models",
|
||||
);
|
||||
|
||||
const res = await fetch(API_ENDPOINT);
|
||||
if (!res.ok) {
|
||||
console.error(`Fetch failed: ${res.status} ${res.statusText}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const parsed = AmbientResponse.safeParse(await res.json());
|
||||
if (!parsed.success) {
|
||||
console.error("Invalid Ambient response:", parsed.error.issues);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const selected = parsed.data.data.filter((m) => ALLOWLIST.has(m.id));
|
||||
const missing = [...ALLOWLIST].filter(
|
||||
(id) => !selected.some((m) => m.id === id),
|
||||
);
|
||||
if (missing.length > 0) {
|
||||
console.error(`Allowlisted models missing from API: ${missing.join(", ")}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
let count = 0;
|
||||
for (const model of selected) {
|
||||
const extendsFrom = EXTENDS_MAP[model.id];
|
||||
if (!extendsFrom) {
|
||||
console.error(`No EXTENDS_MAP entry for ${model.id}; skipping`);
|
||||
continue;
|
||||
}
|
||||
const filePath = path.join(outDir, `${model.id}.toml`);
|
||||
const toml = formatToml(model, extendsFrom);
|
||||
if (dryRun) {
|
||||
console.log(`--- ${path.relative(process.cwd(), filePath)} ---`);
|
||||
console.log(toml);
|
||||
} else {
|
||||
await mkdir(path.dirname(filePath), { recursive: true });
|
||||
await Bun.write(filePath, toml);
|
||||
}
|
||||
count++;
|
||||
}
|
||||
|
||||
console.log(
|
||||
`${dryRun ? "Previewed" : "Wrote"} ${count} model file(s) under providers/ambient/models/`,
|
||||
);
|
||||
}
|
||||
|
||||
await main();
|
||||
@@ -0,0 +1,589 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
/**
|
||||
* Generates Chutes model TOML files from the Chutes LLM API.
|
||||
*
|
||||
* Flags:
|
||||
* --dry-run: Preview changes without writing files
|
||||
* --new-only: Only create new models, skip updating existing ones
|
||||
* --keep-orphans: Don't delete TOML files for models no longer in the API
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { mkdir } from "node:fs/promises";
|
||||
import { ModelFamilyValues } from "../src/family.js";
|
||||
|
||||
const API_ENDPOINT = "https://llm.chutes.ai/v1/models";
|
||||
|
||||
enum SkipZeroFields {
|
||||
LimitContext = "limit.context",
|
||||
LimitOutput = "limit.output",
|
||||
}
|
||||
|
||||
const Pricing = z.object({
|
||||
prompt: z.number().optional(),
|
||||
completion: z.number().optional(),
|
||||
input_cache_read: z.number().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const ChutesModel = z.object({
|
||||
id: z.string(),
|
||||
created: z.number(),
|
||||
pricing: Pricing.optional(),
|
||||
context_length: z.number().optional(),
|
||||
max_output_length: z.number().optional(),
|
||||
max_model_len: z.number().optional(),
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
supported_features: z.array(z.string()).optional(),
|
||||
supported_sampling_parameters: z.array(z.string()).optional(),
|
||||
quantization: z.string().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const ChutesResponse = z.object({
|
||||
data: z.array(ChutesModel),
|
||||
}).passthrough();
|
||||
|
||||
interface ExistingModel {
|
||||
name?: string;
|
||||
family?: string;
|
||||
attachment?: boolean;
|
||||
reasoning?: boolean;
|
||||
tool_call?: boolean;
|
||||
structured_output?: boolean;
|
||||
temperature?: boolean;
|
||||
knowledge?: string;
|
||||
release_date?: string;
|
||||
last_updated?: string;
|
||||
open_weights?: boolean;
|
||||
interleaved?: boolean | { field: string };
|
||||
status?: string;
|
||||
cost?: {
|
||||
input?: number;
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
};
|
||||
limit?: {
|
||||
context?: number;
|
||||
output?: number;
|
||||
};
|
||||
modalities?: {
|
||||
input?: string[];
|
||||
output?: string[];
|
||||
};
|
||||
}
|
||||
|
||||
interface MergedModel {
|
||||
name: string;
|
||||
family?: string;
|
||||
attachment: boolean;
|
||||
reasoning: boolean;
|
||||
tool_call: boolean;
|
||||
structured_output?: boolean;
|
||||
temperature: boolean;
|
||||
knowledge?: string;
|
||||
release_date: string;
|
||||
last_updated: string;
|
||||
open_weights: boolean;
|
||||
interleaved?: boolean | { field: string };
|
||||
status?: string;
|
||||
cost?: {
|
||||
input: number;
|
||||
output: number;
|
||||
cache_read?: number;
|
||||
};
|
||||
limit: {
|
||||
context: number;
|
||||
output: number;
|
||||
};
|
||||
modalities: {
|
||||
input: string[];
|
||||
output: string[];
|
||||
};
|
||||
}
|
||||
|
||||
interface Changes {
|
||||
field: string;
|
||||
oldValue: string;
|
||||
newValue: string;
|
||||
}
|
||||
|
||||
// ── Utility functions ────────────────────────────────────────────────
|
||||
|
||||
function timestampToDate(timestamp: number): string {
|
||||
const date = new Date(timestamp * 1000);
|
||||
return date.toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function getTodayDate(): string {
|
||||
return new Date().toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function formatNumber(n: number): string {
|
||||
if (n >= 1000) {
|
||||
return n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_");
|
||||
}
|
||||
return n.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Humanize a model ID into a readable name.
|
||||
* Strips the org prefix and replaces hyphens with spaces.
|
||||
* e.g. "Qwen/Qwen3-32B-TEE" → "Qwen3 32B TEE"
|
||||
*/
|
||||
function humanizeModelName(modelId: string): string {
|
||||
const parts = modelId.split("/");
|
||||
const modelPart = parts[parts.length - 1];
|
||||
return modelPart.replace(/-/g, " ");
|
||||
}
|
||||
|
||||
// ── Family inference ───────────
|
||||
|
||||
function isSubstring(target: string, family: string): boolean {
|
||||
return target.toLowerCase().includes(family.toLowerCase());
|
||||
}
|
||||
|
||||
function matchesFamily(target: string, family: string): boolean {
|
||||
const targetLower = target.toLowerCase();
|
||||
const familyLower = family.toLowerCase();
|
||||
let familyIdx = 0;
|
||||
|
||||
for (let i = 0; i < targetLower.length && familyIdx < familyLower.length; i++) {
|
||||
if (targetLower[i] === familyLower[familyIdx]) {
|
||||
familyIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
return familyIdx === familyLower.length;
|
||||
}
|
||||
|
||||
function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
const sortedFamilies = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
|
||||
|
||||
// First pass: try exact substring matches
|
||||
for (const family of sortedFamilies) {
|
||||
if (isSubstring(modelId, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
if (isSubstring(modelName, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
// Second pass: fall back to subsequence matching
|
||||
for (const family of sortedFamilies) {
|
||||
if (matchesFamily(modelId, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
if (matchesFamily(modelName, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// ── Load existing TOML ───────────────────────────────────────────────
|
||||
|
||||
async function loadExistingModel(filePath: string): Promise<ExistingModel | null> {
|
||||
try {
|
||||
const file = Bun.file(filePath);
|
||||
if (!(await file.exists())) {
|
||||
return null;
|
||||
}
|
||||
const toml = await import(filePath, { with: { type: "toml" } }).then(
|
||||
(mod) => mod.default,
|
||||
);
|
||||
return toml as ExistingModel;
|
||||
} catch (e) {
|
||||
console.warn(`Warning: Failed to parse existing file ${filePath}:`, e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// ── Merge API data with existing TOML ────────────────────────────────
|
||||
|
||||
function mergeModel(
|
||||
apiModel: z.infer<typeof ChutesModel>,
|
||||
existing: ExistingModel | null,
|
||||
): MergedModel {
|
||||
const features = new Set(apiModel.supported_features ?? []);
|
||||
const samplingParams = new Set(apiModel.supported_sampling_parameters ?? []);
|
||||
const inputMods = apiModel.input_modalities ?? ["text"];
|
||||
const outputMods = apiModel.output_modalities ?? ["text"];
|
||||
|
||||
// Capabilities from API features
|
||||
const hasAttachment = inputMods.some((m) =>
|
||||
m === "image" || m === "video" || m === "pdf",
|
||||
);
|
||||
const hasReasoning = features.has("reasoning");
|
||||
const hasToolCall = features.has("tools");
|
||||
const hasStructuredOutput = features.has("structured_outputs");
|
||||
const hasTemperature = samplingParams.size > 0
|
||||
? samplingParams.has("temperature")
|
||||
: true; // default true if no sampling params info
|
||||
|
||||
// Preserve existing values when available (manually specified)
|
||||
const modelName = existing?.name ?? humanizeModelName(apiModel.id);
|
||||
const family = existing?.family ?? inferFamily(apiModel.id, modelName);
|
||||
const knowledge = existing?.knowledge;
|
||||
const interleaved = existing?.interleaved;
|
||||
const status = existing?.status;
|
||||
|
||||
// Release date: existing > API created timestamp > today
|
||||
const releaseDate = existing?.release_date
|
||||
?? timestampToDate(apiModel.created)
|
||||
?? getTodayDate();
|
||||
|
||||
// Context limit: prefer context_length, fallback to max_model_len
|
||||
const apiContext = apiModel.context_length ?? apiModel.max_model_len ?? 0;
|
||||
const contextLimit = apiContext > 0
|
||||
? apiContext
|
||||
: (existing?.limit?.context ?? 0);
|
||||
|
||||
// Output limit: prefer max_output_length, fallback to existing
|
||||
const apiOutput = apiModel.max_output_length ?? 0;
|
||||
const outputLimit = apiOutput > 0
|
||||
? apiOutput
|
||||
: (existing?.limit?.output ?? 0);
|
||||
|
||||
const merged: MergedModel = {
|
||||
name: modelName,
|
||||
family,
|
||||
attachment: hasAttachment,
|
||||
reasoning: hasReasoning,
|
||||
tool_call: hasToolCall,
|
||||
temperature: hasTemperature,
|
||||
release_date: releaseDate,
|
||||
last_updated: getTodayDate(),
|
||||
open_weights: true, // Chutes hosts open-weight models
|
||||
...(hasStructuredOutput && { structured_output: hasStructuredOutput }),
|
||||
...(knowledge && { knowledge }),
|
||||
...(interleaved !== undefined && { interleaved }),
|
||||
...(status && { status }),
|
||||
limit: {
|
||||
context: contextLimit,
|
||||
output: outputLimit,
|
||||
},
|
||||
modalities: {
|
||||
input: inputMods,
|
||||
output: outputMods,
|
||||
},
|
||||
};
|
||||
|
||||
// Cost: API values are already in USD per 1M tokens — use directly
|
||||
if (apiModel.pricing) {
|
||||
const inputPrice = apiModel.pricing.prompt;
|
||||
const outputPrice = apiModel.pricing.completion;
|
||||
const cacheReadPrice = apiModel.pricing.input_cache_read;
|
||||
|
||||
if (inputPrice !== undefined && outputPrice !== undefined) {
|
||||
merged.cost = {
|
||||
input: inputPrice,
|
||||
output: outputPrice,
|
||||
...(cacheReadPrice !== undefined && { cache_read: cacheReadPrice }),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return merged;
|
||||
}
|
||||
|
||||
// ── TOML formatting ──────────────────────────────────────────────────
|
||||
|
||||
function formatToml(model: MergedModel): string {
|
||||
const lines: string[] = [];
|
||||
|
||||
lines.push(`# Auto-generated by generate-chutes.ts — do not edit pricing, limits, or capabilities.`);
|
||||
lines.push(`# Manual overrides preserved on re-run: name, family, knowledge, interleaved, status`);
|
||||
lines.push(`name = "${model.name.replace(/"/g, '\\"')}"`);
|
||||
if (model.family) {
|
||||
lines.push(`family = "${model.family}"`);
|
||||
}
|
||||
lines.push(`release_date = "${model.release_date}"`);
|
||||
lines.push(`last_updated = "${model.last_updated}"`);
|
||||
lines.push(`attachment = ${model.attachment}`);
|
||||
lines.push(`reasoning = ${model.reasoning}`);
|
||||
lines.push(`temperature = ${model.temperature}`);
|
||||
lines.push(`tool_call = ${model.tool_call}`);
|
||||
if (model.structured_output !== undefined) {
|
||||
lines.push(`structured_output = ${model.structured_output}`);
|
||||
}
|
||||
lines.push(`open_weights = ${model.open_weights}`);
|
||||
if (model.knowledge) {
|
||||
lines.push(`knowledge = "${model.knowledge}"`);
|
||||
}
|
||||
if (model.status) {
|
||||
lines.push(`status = "${model.status}"`);
|
||||
}
|
||||
|
||||
if (model.cost) {
|
||||
lines.push("");
|
||||
lines.push(`[cost]`);
|
||||
lines.push(`input = ${model.cost.input}`);
|
||||
lines.push(`output = ${model.cost.output}`);
|
||||
if (model.cost.cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${model.cost.cache_read}`);
|
||||
}
|
||||
}
|
||||
|
||||
lines.push("");
|
||||
lines.push(`[limit]`);
|
||||
lines.push(`context = ${formatNumber(model.limit.context)}`);
|
||||
lines.push(`output = ${formatNumber(model.limit.output)}`);
|
||||
|
||||
lines.push("");
|
||||
lines.push(`[modalities]`);
|
||||
lines.push(`input = [${model.modalities.input.map((m) => `"${m}"`).join(", ")}]`);
|
||||
lines.push(`output = [${model.modalities.output.map((m) => `"${m}"`).join(", ")}]`);
|
||||
|
||||
if (model.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (model.interleaved === true) {
|
||||
lines.push(`interleaved = true`);
|
||||
} else if (typeof model.interleaved === "object") {
|
||||
lines.push(`[interleaved]`);
|
||||
lines.push(`field = "${model.interleaved.field}"`);
|
||||
}
|
||||
}
|
||||
|
||||
return lines.join("\n") + "\n";
|
||||
}
|
||||
|
||||
// ── Change detection ─────────────────────────────────────────────────
|
||||
|
||||
function detectChanges(
|
||||
existing: ExistingModel | null,
|
||||
merged: MergedModel,
|
||||
): Changes[] {
|
||||
if (!existing) return [];
|
||||
|
||||
const changes: Changes[] = [];
|
||||
const EPSILON = 0.001;
|
||||
|
||||
const shouldSkipZero = (field: string, oldVal: unknown, newVal: unknown): boolean => {
|
||||
if (!Object.values(SkipZeroFields).includes(field as SkipZeroFields)) {
|
||||
return false;
|
||||
}
|
||||
return (typeof oldVal === "number" && oldVal === 0) || (typeof newVal === "number" && newVal === 0);
|
||||
};
|
||||
|
||||
const formatValue = (val: unknown): string => {
|
||||
if (typeof val === "number") return formatNumber(val);
|
||||
if (Array.isArray(val)) return `[${val.join(", ")}]`;
|
||||
if (val === undefined) return "(none)";
|
||||
return String(val);
|
||||
};
|
||||
|
||||
const isMaterialPriceDiff = (oldPrice: unknown, newPrice: unknown): boolean => {
|
||||
if (oldPrice === 0 && newPrice === undefined) return false;
|
||||
if (oldPrice !== undefined && newPrice !== undefined) {
|
||||
return Math.abs((oldPrice as number) - (newPrice as number)) > EPSILON;
|
||||
}
|
||||
return oldPrice !== newPrice;
|
||||
};
|
||||
|
||||
const compare = (field: string, oldVal: unknown, newVal: unknown) => {
|
||||
if (shouldSkipZero(field, oldVal, newVal)) return;
|
||||
|
||||
const isDiff = field.startsWith("cost.")
|
||||
? isMaterialPriceDiff(oldVal, newVal)
|
||||
: JSON.stringify(oldVal) !== JSON.stringify(newVal);
|
||||
|
||||
if (isDiff) {
|
||||
changes.push({
|
||||
field,
|
||||
oldValue: formatValue(oldVal),
|
||||
newValue: formatValue(newVal),
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
compare("name", existing.name, merged.name);
|
||||
compare("family", existing.family, merged.family);
|
||||
compare("attachment", existing.attachment, merged.attachment);
|
||||
compare("reasoning", existing.reasoning, merged.reasoning);
|
||||
compare("tool_call", existing.tool_call, merged.tool_call);
|
||||
compare("structured_output", existing.structured_output, merged.structured_output);
|
||||
compare("open_weights", existing.open_weights, merged.open_weights);
|
||||
compare("release_date", existing.release_date, merged.release_date);
|
||||
compare("cost.input", existing.cost?.input, merged.cost?.input);
|
||||
compare("cost.output", existing.cost?.output, merged.cost?.output);
|
||||
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
|
||||
compare("limit.context", existing.limit?.context, merged.limit.context);
|
||||
compare("limit.output", existing.limit?.output, merged.limit.output);
|
||||
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
|
||||
compare("modalities.output", existing.modalities?.output, merged.modalities.output);
|
||||
|
||||
return changes;
|
||||
}
|
||||
|
||||
// ── Main ─────────────────────────────────────────────────────────────
|
||||
|
||||
async function main() {
|
||||
const args = process.argv.slice(2);
|
||||
const dryRun = args.includes("--dry-run");
|
||||
const newOnly = args.includes("--new-only");
|
||||
const keepOrphans = args.includes("--keep-orphans");
|
||||
|
||||
const modelsDir = path.join(
|
||||
import.meta.dirname,
|
||||
"..",
|
||||
"..",
|
||||
"..",
|
||||
"providers",
|
||||
"chutes",
|
||||
"models",
|
||||
);
|
||||
|
||||
console.log(`${dryRun ? "[DRY RUN] " : ""}${newOnly ? "[NEW ONLY] " : ""}${keepOrphans ? "[KEEP ORPHANS] " : ""}Fetching Chutes models from API...`);
|
||||
|
||||
const res = await fetch(API_ENDPOINT);
|
||||
if (!res.ok) {
|
||||
console.error(`Failed to fetch API: ${res.status} ${res.statusText}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const json = await res.json();
|
||||
const parsed = ChutesResponse.safeParse(json);
|
||||
if (!parsed.success) {
|
||||
console.error("Invalid API response:", parsed.error.errors);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const apiModels = parsed.data.data;
|
||||
|
||||
// Scan existing TOML files
|
||||
const existingFiles = new Set<string>();
|
||||
try {
|
||||
for await (const file of new Bun.Glob("**/*.toml").scan({
|
||||
cwd: modelsDir,
|
||||
absolute: false,
|
||||
})) {
|
||||
existingFiles.add(file);
|
||||
}
|
||||
} catch {
|
||||
}
|
||||
|
||||
console.log(`Found ${apiModels.length} models in API, ${existingFiles.size} existing files\n`);
|
||||
|
||||
const apiModelIds = new Set<string>();
|
||||
|
||||
let created = 0;
|
||||
let updated = 0;
|
||||
let unchanged = 0;
|
||||
|
||||
for (const apiModel of apiModels) {
|
||||
const relativePath = `${apiModel.id}.toml`;
|
||||
const filePath = path.join(modelsDir, relativePath);
|
||||
const dirPath = path.dirname(filePath);
|
||||
|
||||
apiModelIds.add(relativePath);
|
||||
|
||||
const existing = await loadExistingModel(filePath);
|
||||
const merged = mergeModel(apiModel, existing);
|
||||
const tomlContent = formatToml(merged);
|
||||
|
||||
if (existing === null) {
|
||||
created++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would create: ${relativePath}`);
|
||||
console.log(` name = "${merged.name}"`);
|
||||
if (merged.family) {
|
||||
console.log(` family = "${merged.family}" (inferred)`);
|
||||
}
|
||||
console.log("");
|
||||
} else {
|
||||
await mkdir(dirPath, { recursive: true });
|
||||
await Bun.write(filePath, tomlContent);
|
||||
console.log(`Created: ${relativePath}`);
|
||||
}
|
||||
} else {
|
||||
if (newOnly) {
|
||||
unchanged++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const changes = detectChanges(existing, merged);
|
||||
const existingContent = await Bun.file(filePath).text();
|
||||
const formatChanged = existingContent !== tomlContent;
|
||||
|
||||
if (changes.length > 0 || formatChanged) {
|
||||
updated++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would update: ${relativePath}`);
|
||||
} else {
|
||||
await mkdir(dirPath, { recursive: true });
|
||||
await Bun.write(filePath, tomlContent);
|
||||
console.log(`Updated: ${relativePath}`);
|
||||
}
|
||||
for (const change of changes) {
|
||||
console.log(` ${change.field}: ${change.oldValue} → ${change.newValue}`);
|
||||
}
|
||||
if (changes.length === 0 && formatChanged) {
|
||||
console.log(` (format-only change)`);
|
||||
}
|
||||
console.log("");
|
||||
} else {
|
||||
unchanged++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Handle orphaned files (on disk but not in API)
|
||||
const orphaned: string[] = [];
|
||||
for (const file of existingFiles) {
|
||||
if (!apiModelIds.has(file)) {
|
||||
orphaned.push(file);
|
||||
const orphanPath = path.join(modelsDir, file);
|
||||
if (keepOrphans) {
|
||||
console.log(`Orphaned (kept): ${file}`);
|
||||
} else if (dryRun) {
|
||||
console.log(`[DRY RUN] Would delete: ${file}`);
|
||||
} else {
|
||||
await Bun.file(orphanPath).delete();
|
||||
console.log(`Deleted: ${file}`);
|
||||
|
||||
// Clean up empty parent directories
|
||||
const parentDir = path.dirname(orphanPath);
|
||||
try {
|
||||
const remaining = [];
|
||||
for await (const entry of new Bun.Glob("*").scan({ cwd: parentDir })) {
|
||||
remaining.push(entry);
|
||||
}
|
||||
if (remaining.length === 0) {
|
||||
const { rmdir } = await import("node:fs/promises");
|
||||
await rmdir(parentDir);
|
||||
console.log(` Removed empty directory: ${path.basename(parentDir)}/`);
|
||||
}
|
||||
} catch {
|
||||
// Directory not empty or other error, ignore
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log("");
|
||||
if (dryRun) {
|
||||
console.log(
|
||||
`Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} would be deleted`,
|
||||
);
|
||||
} else if (keepOrphans) {
|
||||
console.log(
|
||||
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned (kept)`,
|
||||
);
|
||||
} else {
|
||||
console.log(
|
||||
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} deleted`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
await main();
|
||||
@@ -0,0 +1,287 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
/**
|
||||
* Generates Databricks model TOML files from the Foundation Model API endpoint.
|
||||
*
|
||||
* Each Databricks endpoint exposes a model from another provider (Anthropic,
|
||||
* OpenAI, Google, etc.), so the generated TOML uses [extends] to inherit
|
||||
* canonical metadata from that upstream provider's TOML in models.dev.
|
||||
*
|
||||
* Usage:
|
||||
* DATABRICKS_HOST=<host> DATABRICKS_TOKEN=<pat> bun run databricks:generate
|
||||
* bun run databricks:generate --workspace <host> --token <pat>
|
||||
*
|
||||
* Flags:
|
||||
* --dry-run: Preview changes without writing files
|
||||
* --new-only: Only create new models, skip updating existing ones
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { mkdir, readFile } from "node:fs/promises";
|
||||
import { existsSync } from "node:fs";
|
||||
|
||||
const args = process.argv.slice(2);
|
||||
const flag = (name: string) => {
|
||||
const i = args.indexOf(`--${name}`);
|
||||
return i !== -1 ? args[i + 1] : undefined;
|
||||
};
|
||||
const dryRun = args.includes("--dry-run");
|
||||
const newOnly = args.includes("--new-only");
|
||||
|
||||
const host = flag("workspace") ?? process.env.DATABRICKS_HOST;
|
||||
const token = flag("token") ?? process.env.DATABRICKS_TOKEN;
|
||||
|
||||
if (!host || !token) {
|
||||
console.error(
|
||||
"Usage: DATABRICKS_HOST=<host> DATABRICKS_TOKEN=<pat> bun run databricks:generate",
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const workspace = host.replace(/^https?:\/\//, "").replace(/\/$/, "");
|
||||
const PROVIDERS_DIR = path.join(import.meta.dirname, "..", "..", "..", "providers");
|
||||
const MODELS_DIR = path.join(PROVIDERS_DIR, "databricks", "models");
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// API schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const FoundationModel = z
|
||||
.object({
|
||||
ai_gateway_v2_supported: z.boolean().optional(),
|
||||
api_types: z.array(z.string()).optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const ServedEntity = z
|
||||
.object({
|
||||
foundation_model: FoundationModel.optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const Endpoint = z
|
||||
.object({
|
||||
name: z.string(),
|
||||
config: z
|
||||
.object({
|
||||
served_entities: z.array(ServedEntity).optional(),
|
||||
})
|
||||
.passthrough()
|
||||
.optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const FoundationModelsResponse = z
|
||||
.object({
|
||||
endpoints: z.array(Endpoint),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Canonical resolution: map a Databricks endpoint name to a models.dev entry
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const PREFIX_TO_PROVIDER: [string, string][] = [
|
||||
["claude-", "anthropic"],
|
||||
["gpt-", "openai"],
|
||||
["gemini-", "google"],
|
||||
["mistral-", "mistral"],
|
||||
["mixtral-", "mistral"],
|
||||
];
|
||||
|
||||
type Resolution =
|
||||
| { type: "extends"; from: string }
|
||||
| { type: "inline"; content: string }
|
||||
| null;
|
||||
|
||||
async function resolveCanonical(endpointName: string): Promise<Resolution> {
|
||||
const bare = endpointName.replace(/^databricks-/, "");
|
||||
|
||||
// Models in provider subdirectories (e.g. openrouter/openai/gpt-oss-*)
|
||||
// can't use [extends] (schema requires provider/model format), so inline.
|
||||
if (bare.startsWith("gpt-oss-")) {
|
||||
const p = path.join(PROVIDERS_DIR, "openrouter", "models", "openai", `${bare}.toml`);
|
||||
if (existsSync(p)) {
|
||||
return { type: "inline", content: await readFile(p, "utf8") };
|
||||
}
|
||||
}
|
||||
|
||||
// Meta Llama: "meta-llama-3-3-70b-instruct" → "llama-3.3-70b-instruct"
|
||||
if (bare.startsWith("meta-llama-") || bare.startsWith("llama-")) {
|
||||
const llamaId = bare
|
||||
.replace(/^meta-llama-/, "llama-")
|
||||
.replace(/^(llama-\d+)-(\d+)-/, "$1.$2-");
|
||||
const p = path.join(PROVIDERS_DIR, "llama", "models", `${llamaId}.toml`);
|
||||
if (existsSync(p)) return { type: "extends", from: `llama/${llamaId}` };
|
||||
}
|
||||
|
||||
for (const [prefix, provider] of PREFIX_TO_PROVIDER) {
|
||||
if (!bare.startsWith(prefix)) continue;
|
||||
|
||||
const exact = path.join(PROVIDERS_DIR, provider, "models", `${bare}.toml`);
|
||||
if (existsSync(exact)) return { type: "extends", from: `${provider}/${bare}` };
|
||||
|
||||
// Try with hyphens-as-dots in version (e.g. gpt-5-4 → gpt-5.4)
|
||||
const dotted = bare.replace(/^((?:[a-z]+-)+\d+)-(\d)/, "$1.$2");
|
||||
if (dotted !== bare) {
|
||||
const dottedExact = path.join(PROVIDERS_DIR, provider, "models", `${dotted}.toml`);
|
||||
if (existsSync(dottedExact)) return { type: "extends", from: `${provider}/${dotted}` };
|
||||
}
|
||||
|
||||
// Fuzzy: longest filename that shares a prefix with bare or its dotted form
|
||||
const candidates = [bare, ...(dotted !== bare ? [dotted] : [])];
|
||||
const files: string[] = [];
|
||||
try {
|
||||
for await (const f of new Bun.Glob("*.toml").scan({
|
||||
cwd: path.join(PROVIDERS_DIR, provider, "models"),
|
||||
})) {
|
||||
files.push(f);
|
||||
}
|
||||
} catch {
|
||||
// provider directory may not exist
|
||||
}
|
||||
const match = files
|
||||
.map((f) => f.replace(/\.toml$/, ""))
|
||||
.filter((id) => candidates.some((c) => id.startsWith(c) || c.startsWith(id)))
|
||||
.sort((a, b) => b.length - a.length)[0];
|
||||
if (match) return { type: "extends", from: `${provider}/${match}` };
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function formatToml(resolution: Resolution, endpointName: string): string {
|
||||
if (resolution?.type === "extends") {
|
||||
return `[extends]\nfrom = "${resolution.from}"\n`;
|
||||
}
|
||||
if (resolution?.type === "inline") {
|
||||
return resolution.content;
|
||||
}
|
||||
return `# TODO: fill in details for ${endpointName}\nname = "${endpointName}"\n`;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Main
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const IGNORE_PREFIXES = [
|
||||
"databricks-llama-",
|
||||
"databricks-meta-llama-",
|
||||
"databricks-qwen",
|
||||
"databricks-gemma-",
|
||||
];
|
||||
|
||||
async function main() {
|
||||
console.log(
|
||||
`${dryRun ? "[DRY RUN] " : ""}${newOnly ? "[NEW ONLY] " : ""}Fetching Databricks foundation-models...`,
|
||||
);
|
||||
|
||||
const url = `https://${workspace}/api/2.0/serving-endpoints:foundation-models`;
|
||||
const res = await fetch(url, {
|
||||
headers: { Authorization: `Bearer ${token}` },
|
||||
});
|
||||
if (!res.ok) {
|
||||
console.error(`Failed to fetch API: ${res.status} ${res.statusText}`);
|
||||
console.error(await res.text().catch(() => ""));
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const json = await res.json();
|
||||
const parsed = FoundationModelsResponse.safeParse(json);
|
||||
if (!parsed.success) {
|
||||
console.error("Invalid API response:", parsed.error.errors);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const endpoints = parsed.data.endpoints.filter(
|
||||
(e) =>
|
||||
!IGNORE_PREFIXES.some((p) => e.name.startsWith(p)) &&
|
||||
e.config?.served_entities?.some(
|
||||
(se) =>
|
||||
se.foundation_model?.ai_gateway_v2_supported === true &&
|
||||
se.foundation_model?.api_types?.includes("mlflow/v1/chat/completions"),
|
||||
),
|
||||
);
|
||||
|
||||
const existingFiles = new Set<string>();
|
||||
try {
|
||||
for await (const f of new Bun.Glob("*.toml").scan({ cwd: MODELS_DIR })) {
|
||||
existingFiles.add(f);
|
||||
}
|
||||
} catch {
|
||||
// directory may not exist yet
|
||||
}
|
||||
|
||||
console.log(
|
||||
`Found ${endpoints.length} models in API, ${existingFiles.size} existing files\n`,
|
||||
);
|
||||
|
||||
const apiModelIds = new Set<string>();
|
||||
let created = 0;
|
||||
let updated = 0;
|
||||
let unchanged = 0;
|
||||
|
||||
for (const ep of endpoints) {
|
||||
const filename = `${ep.name}.toml`;
|
||||
apiModelIds.add(filename);
|
||||
const filePath = path.join(MODELS_DIR, filename);
|
||||
|
||||
const resolution = await resolveCanonical(ep.name);
|
||||
const newContent = formatToml(resolution, ep.name);
|
||||
const tag = resolution?.type === "extends" ? `extends ${resolution.from}` : resolution?.type ?? "stub";
|
||||
|
||||
const existed = existsSync(filePath);
|
||||
if (!existed) {
|
||||
created++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would create: ${filename} → ${tag}`);
|
||||
} else {
|
||||
await mkdir(MODELS_DIR, { recursive: true });
|
||||
await Bun.write(filePath, newContent);
|
||||
console.log(`Created: ${filename} → ${tag}`);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (newOnly) {
|
||||
unchanged++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const existingContent = await readFile(filePath, "utf8");
|
||||
if (existingContent === newContent) {
|
||||
unchanged++;
|
||||
continue;
|
||||
}
|
||||
|
||||
updated++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would update: ${filename} → ${tag}`);
|
||||
} else {
|
||||
await Bun.write(filePath, newContent);
|
||||
console.log(`Updated: ${filename} → ${tag}`);
|
||||
}
|
||||
}
|
||||
|
||||
const orphaned: string[] = [];
|
||||
for (const file of existingFiles) {
|
||||
if (!apiModelIds.has(file)) {
|
||||
orphaned.push(file);
|
||||
console.log(`Warning: Orphaned file (not in API): ${file}`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log("");
|
||||
if (dryRun) {
|
||||
console.log(
|
||||
`Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} orphaned`,
|
||||
);
|
||||
} else {
|
||||
console.log(
|
||||
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
await main();
|
||||
@@ -209,7 +209,18 @@ interface ExistingModel {
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_min?: number;
|
||||
};
|
||||
tiers?: Array<{
|
||||
tier: {
|
||||
type?: "context";
|
||||
size: number;
|
||||
};
|
||||
input?: number;
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
}>;
|
||||
};
|
||||
limit?: {
|
||||
context?: number;
|
||||
@@ -262,6 +273,7 @@ interface MergedModel {
|
||||
output: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_min?: number;
|
||||
};
|
||||
};
|
||||
limit: {
|
||||
@@ -310,6 +322,35 @@ function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function getExistingLongContextCost(existing: ExistingModel | null) {
|
||||
const tier = existing?.cost?.tiers?.find(
|
||||
(tier) =>
|
||||
(tier.tier.type === undefined || tier.tier.type === "context") &&
|
||||
tier.tier.size >= 200_000,
|
||||
);
|
||||
if (tier) {
|
||||
return {
|
||||
...tier,
|
||||
context_min: tier.tier.size,
|
||||
};
|
||||
}
|
||||
|
||||
return existing?.cost?.context_over_200k === undefined
|
||||
? undefined
|
||||
: {
|
||||
...existing.cost.context_over_200k,
|
||||
context_min: 200_000,
|
||||
};
|
||||
}
|
||||
|
||||
function getLongContextMin(cost: { context_min?: number }) {
|
||||
return cost.context_min ?? 200_000;
|
||||
}
|
||||
|
||||
function formatInlineNumber(n: number): string {
|
||||
return n >= 1000 ? n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_") : n.toString();
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Merge API data with existing TOML
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -394,27 +435,30 @@ function mergeModel(
|
||||
};
|
||||
|
||||
// Context-tiered pricing (>200k) from the static-content API
|
||||
const existingLongContextCost = getExistingLongContextCost(existing);
|
||||
if (pricing?.inputOver200k !== undefined && pricing?.outputOver200k !== undefined) {
|
||||
merged.cost.context_over_200k = {
|
||||
input: pricing.inputOver200k,
|
||||
output: pricing.outputOver200k,
|
||||
...(existing?.cost?.context_over_200k?.cache_read !== undefined && {
|
||||
cache_read: existing.cost.context_over_200k.cache_read,
|
||||
context_min: existingLongContextCost?.context_min ?? 200_000,
|
||||
...(existingLongContextCost?.cache_read !== undefined && {
|
||||
cache_read: existingLongContextCost.cache_read,
|
||||
}),
|
||||
...(existing?.cost?.context_over_200k?.cache_write !== undefined && {
|
||||
cache_write: existing.cost.context_over_200k.cache_write,
|
||||
...(existingLongContextCost?.cache_write !== undefined && {
|
||||
cache_write: existingLongContextCost.cache_write,
|
||||
}),
|
||||
};
|
||||
} else if (existing?.cost?.context_over_200k) {
|
||||
// Preserve manually-entered context_over_200k if API has no data
|
||||
} else if (existingLongContextCost) {
|
||||
// Preserve manually-entered tiered pricing if API has no data
|
||||
merged.cost.context_over_200k = {
|
||||
input: existing.cost.context_over_200k.input ?? inputPrice,
|
||||
output: existing.cost.context_over_200k.output ?? outputPrice,
|
||||
...(existing.cost.context_over_200k.cache_read !== undefined && {
|
||||
cache_read: existing.cost.context_over_200k.cache_read,
|
||||
input: existingLongContextCost.input ?? inputPrice,
|
||||
output: existingLongContextCost.output ?? outputPrice,
|
||||
context_min: existingLongContextCost.context_min,
|
||||
...(existingLongContextCost.cache_read !== undefined && {
|
||||
cache_read: existingLongContextCost.cache_read,
|
||||
}),
|
||||
...(existing.cost.context_over_200k.cache_write !== undefined && {
|
||||
cache_write: existing.cost.context_over_200k.cache_write,
|
||||
...(existingLongContextCost.cache_write !== undefined && {
|
||||
cache_write: existingLongContextCost.cache_write,
|
||||
}),
|
||||
};
|
||||
}
|
||||
@@ -463,7 +507,8 @@ function formatToml(model: MergedModel): string {
|
||||
|
||||
if (model.cost.context_over_200k) {
|
||||
lines.push("");
|
||||
lines.push(`[cost.context_over_200k]`);
|
||||
lines.push(`[[cost.tiers]]`);
|
||||
lines.push(`tier = { size = ${formatInlineNumber(getLongContextMin(model.cost.context_over_200k))} }`);
|
||||
lines.push(`input = ${model.cost.context_over_200k.input}`);
|
||||
lines.push(`output = ${model.cost.context_over_200k.output}`);
|
||||
if (model.cost.context_over_200k.cache_read !== undefined)
|
||||
@@ -524,8 +569,9 @@ function detectChanges(existing: ExistingModel | null, merged: MergedModel): Cha
|
||||
compare("status", existing.status, merged.status);
|
||||
compare("cost.input", existing.cost?.input, merged.cost?.input);
|
||||
compare("cost.output", existing.cost?.output, merged.cost?.output);
|
||||
compare("cost.context_over_200k.input", existing.cost?.context_over_200k?.input, merged.cost?.context_over_200k?.input);
|
||||
compare("cost.context_over_200k.output", existing.cost?.context_over_200k?.output, merged.cost?.context_over_200k?.output);
|
||||
const existingLongContextCost = getExistingLongContextCost(existing);
|
||||
compare("cost.context_over_200k.input", existingLongContextCost?.input, merged.cost?.context_over_200k?.input);
|
||||
compare("cost.context_over_200k.output", existingLongContextCost?.output, merged.cost?.context_over_200k?.output);
|
||||
compare("limit.context", existing.limit?.context, merged.limit.context);
|
||||
compare("limit.output", existing.limit?.output, merged.limit.output);
|
||||
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
|
||||
|
||||
@@ -162,7 +162,18 @@ interface ExistingModel {
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_min?: number;
|
||||
};
|
||||
tiers?: Array<{
|
||||
tier: {
|
||||
type?: "context";
|
||||
size: number;
|
||||
};
|
||||
input?: number;
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
}>;
|
||||
};
|
||||
limit?: {
|
||||
context?: number;
|
||||
@@ -195,6 +206,30 @@ async function loadExistingModel(filePath: string): Promise<ExistingModel | null
|
||||
}
|
||||
}
|
||||
|
||||
function getExistingLongContextMin(existing: ExistingModel | null) {
|
||||
return (
|
||||
existing?.cost?.tiers?.find(
|
||||
(tier) =>
|
||||
(tier.tier.type === undefined || tier.tier.type === "context") &&
|
||||
tier.tier.size >= 200_000,
|
||||
)?.tier.size ?? 200_000
|
||||
);
|
||||
}
|
||||
|
||||
function getExistingLongContextCost(existing: ExistingModel | null) {
|
||||
return (
|
||||
existing?.cost?.tiers?.find(
|
||||
(tier) =>
|
||||
(tier.tier.type === undefined || tier.tier.type === "context") &&
|
||||
tier.tier.size >= 200_000,
|
||||
) ?? existing?.cost?.context_over_200k
|
||||
);
|
||||
}
|
||||
|
||||
function getLongContextMin(cost: { context_min?: number }) {
|
||||
return cost.context_min ?? 200_000;
|
||||
}
|
||||
|
||||
interface MergedModel {
|
||||
name: string;
|
||||
family?: string;
|
||||
@@ -219,6 +254,7 @@ interface MergedModel {
|
||||
output: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_min?: number;
|
||||
};
|
||||
};
|
||||
limit: {
|
||||
@@ -292,6 +328,7 @@ function mergeModel(
|
||||
merged.cost.context_over_200k = {
|
||||
input: spec.pricing.extended.input.usd,
|
||||
output: spec.pricing.extended.output.usd,
|
||||
context_min: spec.pricing.extended.context_token_threshold,
|
||||
...(spec.pricing.extended.cache_input && { cache_read: spec.pricing.extended.cache_input.usd }),
|
||||
...(spec.pricing.extended.cache_write && { cache_write: spec.pricing.extended.cache_write.usd }),
|
||||
};
|
||||
@@ -366,7 +403,8 @@ function formatToml(model: MergedModel): string {
|
||||
|
||||
if (model.cost.context_over_200k) {
|
||||
lines.push("");
|
||||
lines.push(`[cost.context_over_200k]`);
|
||||
lines.push(`[[cost.tiers]]`);
|
||||
lines.push(`tier = { size = ${formatNumber(getLongContextMin(model.cost.context_over_200k))} }`);
|
||||
lines.push(`input = ${model.cost.context_over_200k.input}`);
|
||||
lines.push(`output = ${model.cost.context_over_200k.output}`);
|
||||
if (model.cost.context_over_200k.cache_read !== undefined) {
|
||||
@@ -438,10 +476,11 @@ function detectChanges(
|
||||
compare("cost.output", existing.cost?.output, merged.cost?.output);
|
||||
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
|
||||
compare("cost.cache_write", existing.cost?.cache_write, merged.cost?.cache_write);
|
||||
compare("cost.context_over_200k.input", existing.cost?.context_over_200k?.input, merged.cost?.context_over_200k?.input);
|
||||
compare("cost.context_over_200k.output", existing.cost?.context_over_200k?.output, merged.cost?.context_over_200k?.output);
|
||||
compare("cost.context_over_200k.cache_read", existing.cost?.context_over_200k?.cache_read, merged.cost?.context_over_200k?.cache_read);
|
||||
compare("cost.context_over_200k.cache_write", existing.cost?.context_over_200k?.cache_write, merged.cost?.context_over_200k?.cache_write);
|
||||
const existingLongContextCost = getExistingLongContextCost(existing);
|
||||
compare("cost.context_over_200k.input", existingLongContextCost?.input, merged.cost?.context_over_200k?.input);
|
||||
compare("cost.context_over_200k.output", existingLongContextCost?.output, merged.cost?.context_over_200k?.output);
|
||||
compare("cost.context_over_200k.cache_read", existingLongContextCost?.cache_read, merged.cost?.context_over_200k?.cache_read);
|
||||
compare("cost.context_over_200k.cache_write", existingLongContextCost?.cache_write, merged.cost?.context_over_200k?.cache_write);
|
||||
compare("limit.context", existing.limit?.context, merged.limit.context);
|
||||
compare("limit.output", existing.limit?.output, merged.limit.output);
|
||||
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
import { main } from "../src/sync/index.js";
|
||||
|
||||
await main();
|
||||
@@ -55,6 +55,7 @@ export const ModelFamilyValues = [
|
||||
"deepseek",
|
||||
"deepseek-thinking",
|
||||
"deepseek-flash",
|
||||
"deepseek-flash-free",
|
||||
"deepseek-flash-think",
|
||||
|
||||
// Microsoft Phi
|
||||
@@ -81,6 +82,7 @@ export const ModelFamilyValues = [
|
||||
|
||||
// xAI Grok
|
||||
"grok",
|
||||
"grok-build",
|
||||
"grok-vision",
|
||||
"grok-beta",
|
||||
|
||||
@@ -289,12 +291,14 @@ export const ModelFamilyValues = [
|
||||
"rnj",
|
||||
|
||||
// Tecent Hy
|
||||
"hy3",
|
||||
"hy3-free",
|
||||
|
||||
// Ling & Ring (InclusionAI)
|
||||
"ling",
|
||||
"ling-flash-free",
|
||||
"ring",
|
||||
"ring-1t-free",
|
||||
|
||||
// Kat Coder
|
||||
"kat-coder",
|
||||
|
||||
@@ -2,9 +2,9 @@ import path from "path";
|
||||
import { mergeDeep } from "remeda";
|
||||
import { z } from "zod";
|
||||
|
||||
import { Provider, Model } from "./schema.js";
|
||||
import { Provider, Model, AuthoredModel, AuthoredModelShape } from "./schema.js";
|
||||
|
||||
const ExtendsModel = Model.sourceType()
|
||||
const ExtendsModel = AuthoredModelShape
|
||||
.partial()
|
||||
.extend({
|
||||
extends: z
|
||||
@@ -71,12 +71,12 @@ export async function generate(directory: string) {
|
||||
});
|
||||
continue;
|
||||
}
|
||||
const model = Model.safeParse(toml);
|
||||
const model = AuthoredModel.safeParse(toml);
|
||||
if (!model.success) {
|
||||
model.error.cause = { modelPath, toml };
|
||||
throw model.error;
|
||||
}
|
||||
provider.data.models[modelID] = model.data;
|
||||
provider.data.models[modelID] = normalizeModelCost(model.data);
|
||||
}
|
||||
result[providerID] = provider.data;
|
||||
}
|
||||
@@ -144,7 +144,7 @@ export async function generate(directory: string) {
|
||||
}
|
||||
}
|
||||
|
||||
const model = Model.safeParse(merged);
|
||||
const model = Model.safeParse(normalizeCost(merged));
|
||||
if (!model.success) {
|
||||
model.error.cause = { modelPath: pendingModel.modelPath, toml: merged };
|
||||
throw model.error;
|
||||
@@ -155,3 +155,51 @@ export async function generate(directory: string) {
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
function normalizeModelCost(model: z.infer<typeof AuthoredModel>): Model {
|
||||
return normalizeCost(model) as Model;
|
||||
}
|
||||
|
||||
function normalizeCost(model: Record<string, unknown>) {
|
||||
const cost = model.cost;
|
||||
if (cost === undefined || cost === null || typeof cost !== "object" || Array.isArray(cost)) {
|
||||
return model;
|
||||
}
|
||||
|
||||
const tiers = (cost as { tiers?: unknown }).tiers;
|
||||
if (!Array.isArray(tiers)) {
|
||||
return model;
|
||||
}
|
||||
|
||||
if (tiers.length !== 1) {
|
||||
return model;
|
||||
}
|
||||
|
||||
const contextOver200k = tiers.find((tier) => {
|
||||
if (tier === null || typeof tier !== "object" || Array.isArray(tier)) return false;
|
||||
const tierConfig = (tier as { tier?: unknown }).tier;
|
||||
if (tierConfig === null || typeof tierConfig !== "object" || Array.isArray(tierConfig)) return false;
|
||||
const type = (tierConfig as { type?: unknown }).type;
|
||||
const size = (tierConfig as { size?: unknown }).size;
|
||||
// context_over_200k is a legacy compatibility field. It intentionally
|
||||
// includes higher thresholds; cost.tiers carries the exact threshold.
|
||||
return (
|
||||
(type === undefined || type === "context") &&
|
||||
typeof size === "number" &&
|
||||
size >= 200_000
|
||||
);
|
||||
});
|
||||
|
||||
if (contextOver200k === undefined) {
|
||||
return model;
|
||||
}
|
||||
|
||||
const { tier: _tier, ...legacyCost } = contextOver200k as Record<string, unknown>;
|
||||
return {
|
||||
...model,
|
||||
cost: {
|
||||
...(cost as Record<string, unknown>),
|
||||
context_over_200k: legacyCost,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
+151
-97
@@ -21,110 +21,164 @@ const JsonValue: z.ZodType<JsonValue> = z.lazy(() =>
|
||||
]),
|
||||
);
|
||||
|
||||
const Cost = z.object({
|
||||
input: z.number().min(0, "Input price cannot be negative"),
|
||||
output: z.number().min(0, "Output price cannot be negative"),
|
||||
reasoning: z.number().min(0, "Input price cannot be negative").optional(),
|
||||
cache_read: z
|
||||
.number()
|
||||
.min(0, "Cache read price cannot be negative")
|
||||
const Cost = z
|
||||
.object({
|
||||
input: z.number().min(0, "Input price cannot be negative"),
|
||||
output: z.number().min(0, "Output price cannot be negative"),
|
||||
reasoning: z
|
||||
.number()
|
||||
.min(0, "Reasoning price cannot be negative")
|
||||
.optional(),
|
||||
cache_read: z
|
||||
.number()
|
||||
.min(0, "Cache read price cannot be negative")
|
||||
.optional(),
|
||||
cache_write: z
|
||||
.number()
|
||||
.min(0, "Cache write price cannot be negative")
|
||||
.optional(),
|
||||
input_audio: z
|
||||
.number()
|
||||
.min(0, "Audio input price cannot be negative")
|
||||
.optional(),
|
||||
output_audio: z
|
||||
.number()
|
||||
.min(0, "Audio output price cannot be negative")
|
||||
.optional(),
|
||||
});
|
||||
|
||||
const CostTier = Cost.extend({
|
||||
tier: z
|
||||
.object({
|
||||
type: z.literal("context").default("context"),
|
||||
size: z.number().int().min(0, "Context tier size cannot be negative"),
|
||||
})
|
||||
.strict(),
|
||||
}).strict();
|
||||
|
||||
const AuthoredCost = Cost.extend({
|
||||
context_over_200k: z.never().optional(),
|
||||
tiers: z.array(CostTier).optional(),
|
||||
});
|
||||
|
||||
const OutputCost = Cost.extend({
|
||||
context_over_200k: Cost.optional(),
|
||||
tiers: z.array(CostTier).optional(),
|
||||
});
|
||||
|
||||
const ModelBase = z.object({
|
||||
id: z.string(),
|
||||
name: z.string().min(1, "Model name cannot be empty"),
|
||||
family: ModelFamily.optional(),
|
||||
attachment: z.boolean(),
|
||||
reasoning: z.boolean(),
|
||||
tool_call: z.boolean(),
|
||||
interleaved: z
|
||||
.union([
|
||||
z.literal(true),
|
||||
z
|
||||
.object({
|
||||
field: z.enum(["reasoning_content", "reasoning_details"]),
|
||||
})
|
||||
.strict(),
|
||||
])
|
||||
.optional(),
|
||||
cache_write: z
|
||||
.number()
|
||||
.min(0, "Cache write price cannot be negative")
|
||||
structured_output: z.boolean().optional(),
|
||||
temperature: z.boolean().optional(),
|
||||
knowledge: z
|
||||
.string()
|
||||
.regex(/^\d{4}-\d{2}(-\d{2})?$/, {
|
||||
message: "Must be in YYYY-MM or YYYY-MM-DD format",
|
||||
})
|
||||
.optional(),
|
||||
input_audio: z
|
||||
.number()
|
||||
.min(0, "Audio input price cannot be negative")
|
||||
release_date: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
|
||||
message: "Must be in YYYY-MM or YYYY-MM-DD format",
|
||||
}),
|
||||
last_updated: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
|
||||
message: "Must be in YYYY-MM or YYYY-MM-DD format",
|
||||
}),
|
||||
modalities: z.object({
|
||||
input: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
|
||||
output: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
|
||||
}),
|
||||
open_weights: z.boolean(),
|
||||
limit: z.object({
|
||||
context: z.number().min(0, "Context window must be positive"),
|
||||
input: z.number().min(0, "Input tokens must be positive").optional(),
|
||||
output: z.number().min(0, "Output tokens must be positive"),
|
||||
}),
|
||||
status: z.enum(["alpha", "beta", "deprecated"]).optional(),
|
||||
experimental: z
|
||||
.object({
|
||||
modes: z
|
||||
.record(
|
||||
z.object({
|
||||
cost: Cost.optional(),
|
||||
provider: z
|
||||
.object({
|
||||
body: z.record(JsonValue).optional(),
|
||||
headers: z.record(z.string()).optional(),
|
||||
})
|
||||
.optional(),
|
||||
}),
|
||||
)
|
||||
.optional(),
|
||||
})
|
||||
.optional(),
|
||||
output_audio: z
|
||||
.number()
|
||||
.min(0, "Audio output price cannot be negative")
|
||||
provider: z
|
||||
.object({
|
||||
npm: z.string().optional(),
|
||||
api: z.string().optional(),
|
||||
shape: z.enum(["responses", "completions"]).optional(),
|
||||
body: z.record(JsonValue).optional(),
|
||||
headers: z.record(z.string()).optional(),
|
||||
})
|
||||
.optional(),
|
||||
});
|
||||
export const Model = z
|
||||
|
||||
function refineModel<T extends z.ZodTypeAny>(schema: T) {
|
||||
return schema
|
||||
.refine(
|
||||
(data) => {
|
||||
return !(data.reasoning === false && data.cost?.reasoning !== undefined);
|
||||
},
|
||||
{
|
||||
message: "Cannot set cost.reasoning when reasoning is false",
|
||||
path: ["cost", "reasoning"],
|
||||
},
|
||||
)
|
||||
.refine(
|
||||
(data) => {
|
||||
const tiers = data.cost?.tiers;
|
||||
if (tiers === undefined) return true;
|
||||
|
||||
const sizes = tiers.map((tier: { tier: { size: number } }) => tier.tier.size);
|
||||
return new Set(sizes).size === sizes.length;
|
||||
},
|
||||
{
|
||||
message: "Cost context tiers must not have duplicate sizes",
|
||||
path: ["cost", "tiers"],
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
export const ModelShape = z
|
||||
.object({
|
||||
id: z.string(),
|
||||
name: z.string().min(1, "Model name cannot be empty"),
|
||||
family: ModelFamily.optional(),
|
||||
attachment: z.boolean(),
|
||||
reasoning: z.boolean(),
|
||||
tool_call: z.boolean(),
|
||||
interleaved: z
|
||||
.union([
|
||||
z.literal(true),
|
||||
z
|
||||
.object({
|
||||
field: z.enum(["reasoning_content", "reasoning_details"]),
|
||||
})
|
||||
.strict(),
|
||||
])
|
||||
.optional(),
|
||||
structured_output: z.boolean().optional(),
|
||||
temperature: z.boolean().optional(),
|
||||
knowledge: z
|
||||
.string()
|
||||
.regex(/^\d{4}-\d{2}(-\d{2})?$/, {
|
||||
message: "Must be in YYYY-MM or YYYY-MM-DD format",
|
||||
})
|
||||
.optional(),
|
||||
release_date: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
|
||||
message: "Must be in YYYY-MM or YYYY-MM-DD format",
|
||||
}),
|
||||
last_updated: z.string().regex(/^\d{4}-\d{2}(-\d{2})?$/, {
|
||||
message: "Must be in YYYY-MM or YYYY-MM-DD format",
|
||||
}),
|
||||
modalities: z.object({
|
||||
input: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
|
||||
output: z.array(z.enum(["text", "audio", "image", "video", "pdf"])),
|
||||
}),
|
||||
open_weights: z.boolean(),
|
||||
cost: Cost.extend({
|
||||
context_over_200k: Cost.optional(),
|
||||
}).optional(),
|
||||
limit: z.object({
|
||||
context: z.number().min(0, "Context window must be positive"),
|
||||
input: z.number().min(0, "Input tokens must be positive").optional(),
|
||||
output: z.number().min(0, "Output tokens must be positive"),
|
||||
}),
|
||||
status: z.enum(["alpha", "beta", "deprecated"]).optional(),
|
||||
experimental: z
|
||||
.object({
|
||||
modes: z
|
||||
.record(
|
||||
z.object({
|
||||
cost: Cost.optional(),
|
||||
provider: z
|
||||
.object({
|
||||
body: z.record(JsonValue).optional(),
|
||||
headers: z.record(z.string()).optional(),
|
||||
})
|
||||
.optional(),
|
||||
}),
|
||||
)
|
||||
.optional(),
|
||||
})
|
||||
.optional(),
|
||||
provider: z
|
||||
.object({
|
||||
npm: z.string().optional(),
|
||||
api: z.string().optional(),
|
||||
shape: z.enum(["responses", "completions"]).optional(),
|
||||
body: z.record(JsonValue).optional(),
|
||||
headers: z.record(z.string()).optional(),
|
||||
})
|
||||
.optional(),
|
||||
...ModelBase.shape,
|
||||
cost: OutputCost.optional(),
|
||||
})
|
||||
.strict()
|
||||
.refine(
|
||||
(data) => {
|
||||
return !(data.reasoning === false && data.cost?.reasoning !== undefined);
|
||||
},
|
||||
{
|
||||
message: "Cannot set cost.reasoning when reasoning is false",
|
||||
path: ["cost", "reasoning"],
|
||||
},
|
||||
);
|
||||
.strict();
|
||||
|
||||
export const AuthoredModelShape = z
|
||||
.object({
|
||||
...ModelBase.shape,
|
||||
cost: AuthoredCost.optional(),
|
||||
})
|
||||
.strict();
|
||||
|
||||
export const Model = refineModel(ModelShape);
|
||||
|
||||
export const AuthoredModel = refineModel(AuthoredModelShape);
|
||||
|
||||
export type Model = z.infer<typeof Model>;
|
||||
|
||||
|
||||
@@ -0,0 +1,481 @@
|
||||
import path from "node:path";
|
||||
import { mkdir, readdir, rm } from "node:fs/promises";
|
||||
import { z } from "zod";
|
||||
|
||||
import { AuthoredModel, AuthoredModelShape } from "../schema.js";
|
||||
import { cloudflareWorkersAi } from "./providers/cloudflare-workers-ai.js";
|
||||
import { google } from "./providers/google.js";
|
||||
import { openrouter } from "./providers/openrouter.js";
|
||||
import { xai } from "./providers/xai.js";
|
||||
|
||||
const ExtendsConfig = z
|
||||
.object({
|
||||
from: z.string(),
|
||||
omit: z.array(z.string()).optional(),
|
||||
})
|
||||
.strict();
|
||||
|
||||
const ExistingExtendsConfig = z
|
||||
.object({
|
||||
from: z.string(),
|
||||
omit: z.array(z.string()).optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const ExistingModel = AuthoredModelShape.partial()
|
||||
.extend({
|
||||
extends: ExistingExtendsConfig.optional(),
|
||||
})
|
||||
.strict();
|
||||
|
||||
const SyncedExtendsModel = AuthoredModelShape.partial()
|
||||
.extend({
|
||||
id: z.string(),
|
||||
extends: ExtendsConfig,
|
||||
})
|
||||
.strict();
|
||||
|
||||
const SyncedAuthoredModel = z.union([AuthoredModel, SyncedExtendsModel]);
|
||||
|
||||
export type ExistingModel = z.infer<typeof ExistingModel>;
|
||||
export type SyncedFullModel = Omit<z.infer<typeof AuthoredModelShape>, "id">;
|
||||
export type SyncedExtendsModel = Omit<z.infer<typeof SyncedExtendsModel>, "id">;
|
||||
export type SyncedModel = SyncedFullModel | SyncedExtendsModel;
|
||||
|
||||
export interface SyncProvider<SourceModel> {
|
||||
id: string;
|
||||
name: string;
|
||||
modelsDir: string;
|
||||
skipCreates?: boolean;
|
||||
sourceID?(model: SourceModel): string;
|
||||
skippedNotice?(ids: string[]): string[];
|
||||
fetchModels(): Promise<unknown>;
|
||||
parseModels(raw: unknown): SourceModel[];
|
||||
translateModel(
|
||||
model: SourceModel,
|
||||
context: { existing(id: string): ExistingModel | undefined },
|
||||
): { id: string; model: SyncedModel } | undefined;
|
||||
}
|
||||
|
||||
export interface SyncResult {
|
||||
id: string;
|
||||
name: string;
|
||||
status: "changed" | "unchanged";
|
||||
created: number;
|
||||
updated: number;
|
||||
deleted: number;
|
||||
unchanged: number;
|
||||
notices: string[];
|
||||
files: Array<{ status: "created" | "updated" | "deleted"; path: string }>;
|
||||
}
|
||||
|
||||
export const providers: {
|
||||
"cloudflare-workers-ai": SyncProvider<any>;
|
||||
google: SyncProvider<any>;
|
||||
openrouter: SyncProvider<any>;
|
||||
xai: SyncProvider<any>;
|
||||
} = {
|
||||
"cloudflare-workers-ai": cloudflareWorkersAi,
|
||||
google,
|
||||
openrouter,
|
||||
xai,
|
||||
};
|
||||
|
||||
export const groups = {
|
||||
aggregators: ["openrouter"],
|
||||
cloudflare: ["cloudflare-workers-ai"],
|
||||
direct: ["google", "xai"],
|
||||
} as const;
|
||||
|
||||
type ProviderID = keyof typeof providers;
|
||||
|
||||
interface SyncOptions {
|
||||
dryRun?: boolean;
|
||||
newOnly?: boolean;
|
||||
}
|
||||
|
||||
export async function syncProviderByID(id: ProviderID, options: SyncOptions = {}) {
|
||||
return syncProvider(providers[id], options);
|
||||
}
|
||||
|
||||
export async function syncProvider<SourceModel>(
|
||||
provider: SyncProvider<SourceModel>,
|
||||
options: SyncOptions = {},
|
||||
): Promise<SyncResult> {
|
||||
console.log(`\nSyncing ${provider.name}...`);
|
||||
|
||||
const existing = await readExisting(provider.modelsDir);
|
||||
const sourceModels = provider.parseModels(await provider.fetchModels());
|
||||
const desired = new Map<string, { model: z.infer<typeof SyncedAuthoredModel>; content: string }>();
|
||||
const skippedRemote: string[] = [];
|
||||
|
||||
for (const sourceModel of sourceModels) {
|
||||
const translated = provider.translateModel(sourceModel, {
|
||||
existing(id) {
|
||||
return existing.get(`${id}.toml`)?.toml;
|
||||
},
|
||||
});
|
||||
if (translated === undefined) {
|
||||
if (provider.skipCreates) skippedRemote.push(provider.sourceID?.(sourceModel) ?? "unknown");
|
||||
continue;
|
||||
}
|
||||
|
||||
const relativePath = `${translated.id}.toml`;
|
||||
if (provider.skipCreates && !existing.has(relativePath)) {
|
||||
skippedRemote.push(translated.id);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (desired.has(relativePath)) {
|
||||
throw new Error(`Duplicate synced model path: ${provider.id}/${relativePath}`);
|
||||
}
|
||||
|
||||
const parsed = SyncedAuthoredModel.safeParse({
|
||||
id: translated.id,
|
||||
...translated.model,
|
||||
});
|
||||
if (!parsed.success) {
|
||||
parsed.error.cause = { provider: provider.id, path: relativePath };
|
||||
throw parsed.error;
|
||||
}
|
||||
|
||||
desired.set(relativePath, {
|
||||
model: parsed.data,
|
||||
content: formatToml(parsed.data),
|
||||
});
|
||||
}
|
||||
|
||||
const files: SyncResult["files"] = [];
|
||||
let unchanged = 0;
|
||||
|
||||
for (const [relativePath, file] of desired) {
|
||||
const filePath = path.join(provider.modelsDir, relativePath);
|
||||
const current = existing.get(relativePath);
|
||||
|
||||
if (current === undefined) {
|
||||
files.push({ status: "created", path: filePath });
|
||||
if (options.dryRun) {
|
||||
console.log(`Would create ${relativePath}`);
|
||||
} else {
|
||||
await mkdir(path.dirname(filePath), { recursive: true });
|
||||
await Bun.write(filePath, file.content);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!sameModel(relativePath, current.toml, file.model)) {
|
||||
if (options.newOnly) {
|
||||
unchanged++;
|
||||
continue;
|
||||
}
|
||||
|
||||
files.push({ status: "updated", path: filePath });
|
||||
if (options.dryRun) {
|
||||
console.log(`Would update ${relativePath}`);
|
||||
} else {
|
||||
if (current.symlink) await rm(filePath, { force: true });
|
||||
await Bun.write(filePath, file.content);
|
||||
}
|
||||
} else {
|
||||
unchanged++;
|
||||
}
|
||||
}
|
||||
|
||||
for (const relativePath of existing.keys()) {
|
||||
if (desired.has(relativePath)) continue;
|
||||
if (options.newOnly) {
|
||||
console.log(`Skipping removal in new-only mode: ${relativePath}`);
|
||||
unchanged++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const filePath = path.join(provider.modelsDir, relativePath);
|
||||
files.push({ status: "deleted", path: filePath });
|
||||
if (options.dryRun) {
|
||||
console.log(`Would remove ${relativePath}`);
|
||||
} else {
|
||||
await rm(filePath, { force: true });
|
||||
}
|
||||
}
|
||||
|
||||
const result = summarize(provider, files, unchanged, provider.skippedNotice?.(skippedRemote) ?? []);
|
||||
console.log(
|
||||
`${options.dryRun ? "Dry run: " : ""}${result.created} created, ${result.updated} updated, ${result.deleted} removed, ${result.unchanged} unchanged`,
|
||||
);
|
||||
return result;
|
||||
}
|
||||
|
||||
export async function syncTargets(target: string, options: SyncOptions = {}) {
|
||||
const ids = target in groups
|
||||
? groups[target as keyof typeof groups]
|
||||
: target in providers
|
||||
? [target as ProviderID]
|
||||
: undefined;
|
||||
|
||||
if (ids === undefined) {
|
||||
throw new Error(`Unknown sync target: ${target}`);
|
||||
}
|
||||
|
||||
const results: SyncResult[] = [];
|
||||
for (const id of ids) {
|
||||
results.push(await syncProviderByID(id as ProviderID, options));
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
export function syncProviderMatrix() {
|
||||
return {
|
||||
include: Object.values(providers).map((provider) => ({
|
||||
provider: provider.id,
|
||||
name: provider.name,
|
||||
})),
|
||||
};
|
||||
}
|
||||
|
||||
async function readExisting(modelsDir: string) {
|
||||
const existing = new Map<string, { text: string; toml: ExistingModel; symlink: boolean }>();
|
||||
|
||||
for (const { file, symlink } of await tomlFiles(modelsDir)) {
|
||||
const text = await Bun.file(path.join(modelsDir, file)).text();
|
||||
const parsed = ExistingModel.safeParse(Bun.TOML.parse(text));
|
||||
if (!parsed.success) {
|
||||
parsed.error.cause = { path: path.join(modelsDir, file) };
|
||||
throw parsed.error;
|
||||
}
|
||||
existing.set(file, { text, toml: parsed.data, symlink });
|
||||
}
|
||||
|
||||
return existing;
|
||||
}
|
||||
|
||||
async function tomlFiles(root: string, dir = "") {
|
||||
const result: Array<{ file: string; symlink: boolean }> = [];
|
||||
|
||||
for (const entry of await readdir(path.join(root, dir), { withFileTypes: true })) {
|
||||
const file = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
result.push(...await tomlFiles(root, file));
|
||||
} else if (entry.name.endsWith(".toml") && (entry.isFile() || entry.isSymbolicLink())) {
|
||||
result.push({ file, symlink: entry.isSymbolicLink() });
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
function summarize(
|
||||
provider: { id: string; name: string },
|
||||
files: SyncResult["files"],
|
||||
unchanged: number,
|
||||
notices: string[],
|
||||
): SyncResult {
|
||||
return {
|
||||
id: provider.id,
|
||||
name: provider.name,
|
||||
status: files.length > 0 ? "changed" : "unchanged",
|
||||
created: files.filter((file) => file.status === "created").length,
|
||||
updated: files.filter((file) => file.status === "updated").length,
|
||||
deleted: files.filter((file) => file.status === "deleted").length,
|
||||
unchanged,
|
||||
notices,
|
||||
files,
|
||||
};
|
||||
}
|
||||
|
||||
function sameModel(
|
||||
relativePath: string,
|
||||
current: ExistingModel,
|
||||
desired: z.infer<typeof SyncedAuthoredModel>,
|
||||
) {
|
||||
const parsed = SyncedAuthoredModel.safeParse({
|
||||
id: relativePath.slice(0, -5),
|
||||
...current,
|
||||
});
|
||||
return parsed.success && stable(parsed.data) === stable(desired);
|
||||
}
|
||||
|
||||
function stable(value: unknown): string {
|
||||
if (Array.isArray(value)) {
|
||||
const items = value.map(stable);
|
||||
const ordered = value.every((item) => item === null || typeof item !== "object")
|
||||
? items.sort()
|
||||
: items;
|
||||
return `[${ordered.join(",")}]`;
|
||||
}
|
||||
if (value !== null && typeof value === "object") {
|
||||
return `{${Object.entries(value)
|
||||
.filter(([, item]) => item !== undefined)
|
||||
.sort(([a], [b]) => a.localeCompare(b))
|
||||
.map(([key, item]) => `${JSON.stringify(key)}:${stable(item)}`)
|
||||
.join(",")}}`;
|
||||
}
|
||||
return JSON.stringify(value);
|
||||
}
|
||||
|
||||
async function writeReport(target: string, results: SyncResult[]) {
|
||||
await mkdir(".sync", { recursive: true });
|
||||
|
||||
const lines = [
|
||||
`Updates model TOMLs for the \`${target}\` sync target.`,
|
||||
"",
|
||||
"| Provider | Status | Created | Updated | Deleted |",
|
||||
"| --- | --- | ---: | ---: | ---: |",
|
||||
];
|
||||
|
||||
for (const result of results) {
|
||||
lines.push(
|
||||
`| ${result.name} | ${result.status} | ${result.created} | ${result.updated} | ${result.deleted} |`,
|
||||
);
|
||||
}
|
||||
|
||||
for (const result of results.filter((item) => item.files.length > 0)) {
|
||||
lines.push("", `<details><summary>${result.name} changed files</summary>`, "");
|
||||
for (const file of result.files) {
|
||||
lines.push(`- ${file.status}: \`${file.path}\``);
|
||||
}
|
||||
lines.push("", "</details>");
|
||||
}
|
||||
|
||||
const noticeResults = results.filter((item) => item.notices.length > 0);
|
||||
if (noticeResults.length > 0) {
|
||||
lines.push("", "## Notices");
|
||||
for (const result of noticeResults) {
|
||||
lines.push("", `### ${result.name}`);
|
||||
for (const notice of result.notices) {
|
||||
lines.push(`- ${notice}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
lines.push("", "This PR was created automatically by the daily model sync workflow.");
|
||||
await Bun.write(".sync/model-sync-report.md", `${lines.join("\n")}\n`);
|
||||
}
|
||||
|
||||
function quote(value: string) {
|
||||
return `"${value.replaceAll("\\", "\\\\").replaceAll('"', '\\"')}"`;
|
||||
}
|
||||
|
||||
function formatInteger(n: number) {
|
||||
return String(n).replace(/\B(?=(\d{3})+(?!\d))/g, "_");
|
||||
}
|
||||
|
||||
function formatNumber(n: number) {
|
||||
return Number.isInteger(n) ? formatInteger(n) : String(n);
|
||||
}
|
||||
|
||||
function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
|
||||
const lines: string[] = [];
|
||||
const extendsLines: string[] = [];
|
||||
|
||||
if ("extends" in model) {
|
||||
extendsLines.push("[extends]");
|
||||
extendsLines.push(`from = ${quote(model.extends.from)}`);
|
||||
if (model.extends.omit !== undefined) {
|
||||
extendsLines.push(`omit = [${model.extends.omit.map(quote).join(", ")}]`);
|
||||
}
|
||||
}
|
||||
|
||||
if (model.name !== undefined) lines.push(`name = ${quote(model.name)}`);
|
||||
if (model.family !== undefined) lines.push(`family = ${quote(model.family)}`);
|
||||
if (model.release_date !== undefined) lines.push(`release_date = ${quote(model.release_date)}`);
|
||||
if (model.last_updated !== undefined) lines.push(`last_updated = ${quote(model.last_updated)}`);
|
||||
if (model.attachment !== undefined) lines.push(`attachment = ${model.attachment}`);
|
||||
if (model.reasoning !== undefined) lines.push(`reasoning = ${model.reasoning}`);
|
||||
if (model.temperature !== undefined) lines.push(`temperature = ${model.temperature}`);
|
||||
if (model.tool_call !== undefined) lines.push(`tool_call = ${model.tool_call}`);
|
||||
if (model.structured_output !== undefined) {
|
||||
lines.push(`structured_output = ${model.structured_output}`);
|
||||
}
|
||||
if (model.knowledge !== undefined) lines.push(`knowledge = ${quote(model.knowledge)}`);
|
||||
if (model.open_weights !== undefined) lines.push(`open_weights = ${model.open_weights}`);
|
||||
if (model.status !== undefined) lines.push(`status = ${quote(model.status)}`);
|
||||
|
||||
if (extendsLines.length > 0) {
|
||||
if (lines.length > 0) lines.push("");
|
||||
lines.push(...extendsLines);
|
||||
}
|
||||
|
||||
if (model.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (model.interleaved === true) {
|
||||
lines.push("interleaved = true");
|
||||
} else {
|
||||
lines.push("[interleaved]");
|
||||
lines.push(`field = ${quote(model.interleaved.field)}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (model.cost !== undefined) {
|
||||
lines.push("", "[cost]");
|
||||
lines.push(`input = ${formatNumber(model.cost.input)}`);
|
||||
lines.push(`output = ${formatNumber(model.cost.output)}`);
|
||||
if (model.cost.reasoning !== undefined) {
|
||||
lines.push(`reasoning = ${formatNumber(model.cost.reasoning)}`);
|
||||
}
|
||||
if (model.cost.cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${formatNumber(model.cost.cache_read)}`);
|
||||
}
|
||||
if (model.cost.cache_write !== undefined) {
|
||||
lines.push(`cache_write = ${formatNumber(model.cost.cache_write)}`);
|
||||
}
|
||||
if (model.cost.input_audio !== undefined) {
|
||||
lines.push(`input_audio = ${formatNumber(model.cost.input_audio)}`);
|
||||
}
|
||||
if (model.cost.output_audio !== undefined) {
|
||||
lines.push(`output_audio = ${formatNumber(model.cost.output_audio)}`);
|
||||
}
|
||||
|
||||
for (const tier of model.cost.tiers ?? []) {
|
||||
lines.push("", "[[cost.tiers]]");
|
||||
lines.push(`tier = { size = ${formatInteger(tier.tier.size)} }`);
|
||||
lines.push(`input = ${formatNumber(tier.input)}`);
|
||||
lines.push(`output = ${formatNumber(tier.output)}`);
|
||||
if (tier.reasoning !== undefined) lines.push(`reasoning = ${formatNumber(tier.reasoning)}`);
|
||||
if (tier.cache_read !== undefined) lines.push(`cache_read = ${formatNumber(tier.cache_read)}`);
|
||||
if (tier.cache_write !== undefined) lines.push(`cache_write = ${formatNumber(tier.cache_write)}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (model.limit !== undefined) {
|
||||
lines.push("", "[limit]");
|
||||
if (model.limit.context !== undefined) lines.push(`context = ${formatInteger(model.limit.context)}`);
|
||||
if (model.limit.input !== undefined) lines.push(`input = ${formatInteger(model.limit.input)}`);
|
||||
if (model.limit.output !== undefined) lines.push(`output = ${formatInteger(model.limit.output)}`);
|
||||
}
|
||||
|
||||
if (model.modalities !== undefined) {
|
||||
lines.push("", "[modalities]");
|
||||
if (model.modalities.input !== undefined) {
|
||||
lines.push(`input = [${model.modalities.input.map(quote).join(", ")}]`);
|
||||
}
|
||||
if (model.modalities.output !== undefined) {
|
||||
lines.push(`output = [${model.modalities.output.map(quote).join(", ")}]`);
|
||||
}
|
||||
}
|
||||
|
||||
return `${lines.join("\n")}\n`;
|
||||
}
|
||||
|
||||
export async function main(args = process.argv.slice(2)) {
|
||||
if (args.includes("--list-providers")) {
|
||||
console.log(JSON.stringify(syncProviderMatrix()));
|
||||
return;
|
||||
}
|
||||
|
||||
const target = args.find((arg) => !arg.startsWith("-")) ?? "aggregators";
|
||||
const results = await syncTargets(target, {
|
||||
dryRun: args.includes("--dry-run"),
|
||||
newOnly: args.includes("--new-only"),
|
||||
});
|
||||
|
||||
await writeReport(target, results);
|
||||
|
||||
console.log("\nSync summary");
|
||||
for (const result of results) {
|
||||
console.log(
|
||||
`${result.name}: ${result.created} created, ${result.updated} updated, ${result.deleted} deleted`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.main) await main();
|
||||
@@ -0,0 +1,176 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider } from "../index.js";
|
||||
import {
|
||||
buildOpenRouterModel,
|
||||
OpenRouterModel,
|
||||
OpenRouterResponse,
|
||||
} from "./openrouter.js";
|
||||
|
||||
const API_BASE = "https://api.cloudflare.com/client/v4/accounts";
|
||||
|
||||
const CloudflareOpenRouterResponse = z.object({
|
||||
result: z.union([OpenRouterResponse, z.array(OpenRouterModel)]).optional(),
|
||||
result_info: z.object({
|
||||
page: z.number().optional(),
|
||||
total_pages: z.number().optional(),
|
||||
}).passthrough().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const CloudflareModel = z.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
created: z.number(),
|
||||
hugging_face_id: z.string().nullable().optional(),
|
||||
context_length: z.number(),
|
||||
max_output_length: z.number().nullable().optional(),
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
pricing: z.object({
|
||||
prompt: z.string(),
|
||||
completion: z.string(),
|
||||
internal_reasoning: z.string().optional(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
}),
|
||||
supported_features: z.array(z.string()).optional(),
|
||||
supported_sampling_parameters: z.array(z.string()).optional(),
|
||||
}).passthrough();
|
||||
|
||||
const CloudflareResponse = z.object({
|
||||
data: z.array(CloudflareModel),
|
||||
}).passthrough();
|
||||
|
||||
type CloudflareModel = z.infer<typeof CloudflareModel>;
|
||||
|
||||
export const cloudflareWorkersAi = {
|
||||
id: "cloudflare-workers-ai",
|
||||
name: "Cloudflare Workers AI",
|
||||
modelsDir: "providers/cloudflare-workers-ai/models",
|
||||
async fetchModels() {
|
||||
const accountID = process.env.CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID;
|
||||
const token = process.env.CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN;
|
||||
if (accountID === undefined || token === undefined) {
|
||||
throw new Error(
|
||||
"Cloudflare Workers AI sync requires CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID and CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN",
|
||||
);
|
||||
}
|
||||
|
||||
const first = await fetchPage(accountID, token, 1);
|
||||
const models = parseCloudflareModels(first);
|
||||
const pageInfo = CloudflareOpenRouterResponse.safeParse(first).success
|
||||
? CloudflareOpenRouterResponse.parse(first).result_info
|
||||
: undefined;
|
||||
|
||||
for (let page = 2; page <= (pageInfo?.total_pages ?? 1); page++) {
|
||||
models.push(...parseCloudflareModels(await fetchPage(accountID, token, page)));
|
||||
}
|
||||
|
||||
return { data: models };
|
||||
},
|
||||
parseModels(raw) {
|
||||
return parseCloudflareModels(raw);
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const normalized = normalizeModel(model);
|
||||
const id = normalized.id.replace(/^workers-ai\//, "");
|
||||
return {
|
||||
id,
|
||||
model: buildWorkersAiModel(normalized, context.existing(id)),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<CloudflareModel>;
|
||||
|
||||
function buildWorkersAiModel(model: z.infer<typeof OpenRouterModel>, existing: ExistingModel | undefined) {
|
||||
const synced = buildOpenRouterModel(model, existing);
|
||||
return {
|
||||
...synced,
|
||||
name: existing?.name ?? synced.name,
|
||||
release_date: existing?.release_date ?? synced.release_date,
|
||||
last_updated: existing?.last_updated ?? synced.last_updated,
|
||||
limit: {
|
||||
...synced.limit,
|
||||
output: existing?.limit?.output ?? synced.limit.output,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
async function fetchPage(accountID: string, token: string, page: number) {
|
||||
const url = new URL(`${API_BASE}/${accountID}/ai/models/search`);
|
||||
url.searchParams.set("format", "openrouter");
|
||||
url.searchParams.set("per_page", "1000");
|
||||
url.searchParams.set("page", String(page));
|
||||
|
||||
const response = await fetch(url, {
|
||||
headers: { Authorization: `Bearer ${token}` },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(
|
||||
`Cloudflare Workers AI models request failed: ${response.status} ${response.statusText}${await responseDetails(response)}`,
|
||||
);
|
||||
}
|
||||
return response.json();
|
||||
}
|
||||
|
||||
function parseCloudflareModels(raw: unknown) {
|
||||
const cloudflare = CloudflareResponse.safeParse(raw);
|
||||
if (cloudflare.success) return cloudflare.data.data;
|
||||
|
||||
const direct = OpenRouterResponse.safeParse(raw);
|
||||
if (direct.success) return direct.data.data;
|
||||
|
||||
const wrapped = CloudflareOpenRouterResponse.parse(raw);
|
||||
if (wrapped.result === undefined) {
|
||||
throw new Error("Cloudflare Workers AI response did not include model data");
|
||||
}
|
||||
return Array.isArray(wrapped.result) ? wrapped.result : wrapped.result.data;
|
||||
}
|
||||
|
||||
function normalizeModel(model: CloudflareModel) {
|
||||
if ("architecture" in model && "top_provider" in model && "supported_parameters" in model) {
|
||||
return OpenRouterModel.parse(model);
|
||||
}
|
||||
|
||||
return OpenRouterModel.parse({
|
||||
id: model.id.startsWith("@cf/") ? model.id : `@cf/${model.id.replace(/^@cf\//, "")}`,
|
||||
name: model.name,
|
||||
created: model.created,
|
||||
hugging_face_id: model.hugging_face_id ?? null,
|
||||
knowledge_cutoff: null,
|
||||
context_length: model.context_length,
|
||||
architecture: {
|
||||
input_modalities: model.input_modalities ?? ["text"],
|
||||
output_modalities: model.output_modalities ?? ["text"],
|
||||
},
|
||||
pricing: model.pricing,
|
||||
top_provider: {
|
||||
context_length: model.context_length,
|
||||
max_completion_tokens: model.max_output_length ?? null,
|
||||
},
|
||||
supported_parameters: [
|
||||
...model.supported_sampling_parameters ?? [],
|
||||
...model.supported_features ?? [],
|
||||
],
|
||||
});
|
||||
}
|
||||
|
||||
async function responseDetails(response: Response) {
|
||||
const text = await response.text();
|
||||
if (text.length === 0) return "";
|
||||
|
||||
try {
|
||||
const body = z.object({
|
||||
errors: z.array(z.object({
|
||||
code: z.union([z.string(), z.number()]).optional(),
|
||||
message: z.string().optional(),
|
||||
}).passthrough()).optional(),
|
||||
}).passthrough().parse(JSON.parse(text));
|
||||
const details = body.errors
|
||||
?.map((error) => [error.code, error.message].filter(Boolean).join(": "))
|
||||
.filter((message) => message.length > 0)
|
||||
.join("; ");
|
||||
return details === undefined || details.length === 0 ? "" : ` (${details})`;
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js";
|
||||
|
||||
const API_ENDPOINT = "https://generativelanguage.googleapis.com/v1beta/models";
|
||||
|
||||
const GoogleModel = z.object({
|
||||
name: z.string(),
|
||||
baseModelId: z.string().optional(),
|
||||
version: z.string().optional(),
|
||||
displayName: z.string().optional(),
|
||||
description: z.string().optional(),
|
||||
inputTokenLimit: z.number().int().nonnegative(),
|
||||
outputTokenLimit: z.number().int().nonnegative(),
|
||||
supportedGenerationMethods: z.array(z.string()).optional(),
|
||||
temperature: z.number().optional(),
|
||||
topP: z.number().optional(),
|
||||
topK: z.number().optional(),
|
||||
maxTemperature: z.number().optional(),
|
||||
thinking: z.boolean().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const GoogleResponse = z.object({
|
||||
models: z.array(GoogleModel).optional(),
|
||||
nextPageToken: z.string().optional(),
|
||||
}).passthrough();
|
||||
|
||||
type GoogleModel = z.infer<typeof GoogleModel>;
|
||||
|
||||
export const google = {
|
||||
id: "google",
|
||||
name: "Google",
|
||||
modelsDir: "providers/google/models",
|
||||
skipCreates: true,
|
||||
sourceID(model) {
|
||||
return model.name.replace(/^models\//, "");
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} Google models returned by the API were not created because the Models API does not provide authoritative modalities, pricing, knowledge cutoff, release date, tool calling, or structured output metadata. Existing models are still updated from API-authoritative fields.`,
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
async fetchModels() {
|
||||
const key = process.env.GOOGLE_API_KEY
|
||||
?? process.env.GEMINI_API_KEY
|
||||
?? process.env.GOOGLE_GENERATIVE_AI_API_KEY;
|
||||
if (key === undefined) {
|
||||
throw new Error("Google sync requires GOOGLE_API_KEY, GEMINI_API_KEY, or GOOGLE_GENERATIVE_AI_API_KEY");
|
||||
}
|
||||
|
||||
const models: GoogleModel[] = [];
|
||||
let pageToken: string | undefined;
|
||||
|
||||
do {
|
||||
const url = new URL(API_ENDPOINT);
|
||||
url.searchParams.set("key", key);
|
||||
url.searchParams.set("pageSize", "1000");
|
||||
if (pageToken !== undefined) url.searchParams.set("pageToken", pageToken);
|
||||
|
||||
const response = await fetch(url);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Google models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const page = GoogleResponse.parse(await response.json());
|
||||
models.push(...page.models ?? []);
|
||||
pageToken = page.nextPageToken;
|
||||
} while (pageToken !== undefined);
|
||||
|
||||
return { models };
|
||||
},
|
||||
parseModels(raw) {
|
||||
return GoogleResponse.parse(raw).models ?? [];
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const id = model.name.replace(/^models\//, "");
|
||||
const existing = context.existing(id);
|
||||
if (existing === undefined) return undefined;
|
||||
|
||||
return {
|
||||
id,
|
||||
model: buildModel(model, existing),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<GoogleModel>;
|
||||
|
||||
function buildModel(model: GoogleModel, existing: ExistingModel): SyncedModel {
|
||||
const name = existing.name;
|
||||
const releaseDate = existing.release_date;
|
||||
const lastUpdated = existing.last_updated;
|
||||
const attachment = existing.attachment;
|
||||
const reasoning = existing.reasoning;
|
||||
const toolCall = existing.tool_call;
|
||||
const openWeights = existing.open_weights;
|
||||
const limit = existing.limit;
|
||||
const modalities = existing.modalities;
|
||||
|
||||
if (
|
||||
name === undefined
|
||||
|| releaseDate === undefined
|
||||
|| lastUpdated === undefined
|
||||
|| attachment === undefined
|
||||
|| reasoning === undefined
|
||||
|| toolCall === undefined
|
||||
|| openWeights === undefined
|
||||
|| limit === undefined
|
||||
|| modalities === undefined
|
||||
) {
|
||||
throw new Error(`Google model ${model.name} has incomplete local TOML metadata required for sync`);
|
||||
}
|
||||
|
||||
return {
|
||||
name: model.displayName ?? name,
|
||||
family: existing.family,
|
||||
release_date: releaseDate,
|
||||
last_updated: lastUpdated,
|
||||
attachment,
|
||||
reasoning: model.thinking ?? reasoning,
|
||||
temperature: model.temperature !== undefined || model.maxTemperature !== undefined
|
||||
? true
|
||||
: existing.temperature,
|
||||
tool_call: toolCall,
|
||||
structured_output: existing.structured_output,
|
||||
knowledge: existing.knowledge,
|
||||
open_weights: openWeights,
|
||||
status: existing.status,
|
||||
interleaved: existing.interleaved,
|
||||
cost: existing.cost,
|
||||
limit: {
|
||||
input: limit.input,
|
||||
context: model.inputTokenLimit,
|
||||
output: model.outputTokenLimit,
|
||||
},
|
||||
modalities,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
import { z } from "zod";
|
||||
import { readFileSync, readdirSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
|
||||
import { ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
|
||||
const API_ENDPOINT = "https://openrouter.ai/api/v1/models";
|
||||
const PROVIDERS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "providers");
|
||||
const modelFilesByProvider = new Map<string, Set<string>>();
|
||||
const canonicalTomlByModel = new Map<string, Record<string, unknown>>();
|
||||
|
||||
const CANONICAL_PROVIDER_PREFIXES = {
|
||||
anthropic: "anthropic",
|
||||
cohere: "cohere",
|
||||
deepseek: "deepseek",
|
||||
google: "google",
|
||||
meta: "llama",
|
||||
"meta-llama": "llama",
|
||||
minimax: "minimax",
|
||||
mistralai: "mistral",
|
||||
moonshotai: "moonshotai",
|
||||
openai: "openai",
|
||||
"x-ai": "xai",
|
||||
xai: "xai",
|
||||
xiaomi: "xiaomi",
|
||||
zai: "zai",
|
||||
"z-ai": "zai",
|
||||
} as const;
|
||||
|
||||
export const OpenRouterModel = z.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
created: z.number(),
|
||||
hugging_face_id: z.string().nullable(),
|
||||
knowledge_cutoff: z.string().nullable(),
|
||||
context_length: z.number(),
|
||||
architecture: z.object({
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
}),
|
||||
pricing: z.object({
|
||||
prompt: z.string(),
|
||||
completion: z.string(),
|
||||
internal_reasoning: z.string().optional(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
}),
|
||||
top_provider: z.object({
|
||||
context_length: z.number().nullable(),
|
||||
max_completion_tokens: z.number().nullable(),
|
||||
}),
|
||||
supported_parameters: z.array(z.string()),
|
||||
});
|
||||
|
||||
export const OpenRouterResponse = z.object({
|
||||
data: z.array(OpenRouterModel),
|
||||
}).passthrough();
|
||||
|
||||
export type OpenRouterModel = z.infer<typeof OpenRouterModel>;
|
||||
|
||||
export const openrouter = {
|
||||
id: "openrouter",
|
||||
name: "OpenRouter",
|
||||
modelsDir: "providers/openrouter/models",
|
||||
async fetchModels() {
|
||||
const headers = process.env.OPENROUTER_API_KEY
|
||||
? { Authorization: `Bearer ${process.env.OPENROUTER_API_KEY}` }
|
||||
: undefined;
|
||||
const response = await fetch(API_ENDPOINT, { headers });
|
||||
if (!response.ok) {
|
||||
throw new Error(`OpenRouter request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return OpenRouterResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildOpenRouterModel(model, context.existing(model.id)),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<OpenRouterModel>;
|
||||
|
||||
function dateFromTimestamp(timestamp: number) {
|
||||
return new Date(timestamp * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function price(value: string | undefined) {
|
||||
if (value === undefined) return undefined;
|
||||
const number = Number(value);
|
||||
return Number.isFinite(number) && number >= 0
|
||||
? Math.round(number * 1_000_000_000_000) / 1_000_000
|
||||
: undefined;
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[], fallback: Modality[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => value.toLowerCase())
|
||||
.map((value) => value === "file" ? "pdf" : value)
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
|
||||
function inferFamily(model: OpenRouterModel, name: string) {
|
||||
const target = `${model.id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
.find((family) => {
|
||||
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
if (family === "o") {
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=\\d|$|[^a-z0-9])`).test(target);
|
||||
}
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
|
||||
});
|
||||
}
|
||||
|
||||
export function buildOpenRouterModel(model: OpenRouterModel, existing: ExistingModel | undefined): SyncedModel {
|
||||
const params = new Set(model.supported_parameters);
|
||||
const name = model.name.replace(/^[^:]+:\s+/, "");
|
||||
const input = modalities(model.architecture.input_modalities, ["text"]);
|
||||
const output = modalities(model.architecture.output_modalities, ["text"]);
|
||||
const prompt = price(model.pricing.prompt);
|
||||
const completion = price(model.pricing.completion);
|
||||
const reasoning = params.has("reasoning") || params.has("include_reasoning");
|
||||
const context = model.top_provider.context_length ?? model.context_length;
|
||||
const family = inferFamily(model, name);
|
||||
const releaseDate = dateFromTimestamp(model.created);
|
||||
const familyValue = existing?.family === "o" && family !== "o"
|
||||
? family
|
||||
: (existing?.family ?? family);
|
||||
const attachment = input.some((value) => value !== "text");
|
||||
const toolCall = params.has("tools") || params.has("tool_choice");
|
||||
const structuredOutput = params.has("structured_outputs");
|
||||
const knowledge = model.knowledge_cutoff?.slice(0, 10) ?? existing?.knowledge;
|
||||
const openWeights = Boolean(model.hugging_face_id);
|
||||
const cost = prompt !== undefined && completion !== undefined
|
||||
? {
|
||||
input: prompt,
|
||||
output: completion,
|
||||
reasoning: reasoning ? price(model.pricing.internal_reasoning) : undefined,
|
||||
cache_read: price(model.pricing.input_cache_read),
|
||||
cache_write: price(model.pricing.input_cache_write),
|
||||
tiers: existing?.cost?.tiers,
|
||||
}
|
||||
: existing?.cost;
|
||||
const limit = {
|
||||
context,
|
||||
input: existing?.limit?.input,
|
||||
output: model.top_provider.max_completion_tokens ?? existing?.limit?.output ?? context,
|
||||
};
|
||||
const canonical = resolveCanonicalModel(model.id);
|
||||
|
||||
if (canonical !== undefined) {
|
||||
return {
|
||||
extends: {
|
||||
from: canonical.from,
|
||||
omit: canonicalOmit(canonical.provider, canonical.modelID, cost, limit),
|
||||
},
|
||||
...canonicalRuntimeOverrides(canonical.provider, canonical.modelID, {
|
||||
name: model.id.endsWith(":free") ? name : undefined,
|
||||
attachment,
|
||||
reasoning,
|
||||
}),
|
||||
temperature: params.has("temperature"),
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
name,
|
||||
family: familyValue,
|
||||
release_date: releaseDate,
|
||||
last_updated: releaseDate,
|
||||
attachment,
|
||||
reasoning,
|
||||
temperature: params.has("temperature"),
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput,
|
||||
knowledge,
|
||||
open_weights: openWeights,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
} satisfies SyncedFullModel;
|
||||
}
|
||||
|
||||
function resolveCanonicalModel(openrouterID: string) {
|
||||
const [prefix, ...modelParts] = openrouterID.split("/");
|
||||
if (prefix === undefined || modelParts.length === 0) return undefined;
|
||||
if (openrouterID.startsWith("~/") || prefix.startsWith("~")) return undefined;
|
||||
|
||||
const provider = CANONICAL_PROVIDER_PREFIXES[prefix as keyof typeof CANONICAL_PROVIDER_PREFIXES];
|
||||
if (provider === undefined) return undefined;
|
||||
|
||||
const modelID = modelParts.join("/").replace(/:free$/, "");
|
||||
const candidates = canonicalCandidates(provider, modelID);
|
||||
const match = candidates.find((candidate) => {
|
||||
return canonicalModelExists(provider, candidate);
|
||||
});
|
||||
|
||||
return match === undefined
|
||||
? undefined
|
||||
: {
|
||||
from: `${provider}/${match}`,
|
||||
provider,
|
||||
modelID: match,
|
||||
};
|
||||
}
|
||||
|
||||
function canonicalModelExists(provider: string, modelID: string) {
|
||||
let files = modelFilesByProvider.get(provider);
|
||||
if (files === undefined) {
|
||||
try {
|
||||
files = new Set(readdirSync(path.join(PROVIDERS_DIR, provider, "models")));
|
||||
} catch {
|
||||
files = new Set();
|
||||
}
|
||||
modelFilesByProvider.set(provider, files);
|
||||
}
|
||||
return files.has(`${modelID}.toml`);
|
||||
}
|
||||
|
||||
function canonicalOmit(
|
||||
provider: string,
|
||||
modelID: string,
|
||||
cost: SyncedFullModel["cost"],
|
||||
limit: SyncedFullModel["limit"],
|
||||
) {
|
||||
const toml = canonicalToml(provider, modelID);
|
||||
const omit = ["provider", "experimental"].filter((key) => toml[key] !== undefined);
|
||||
|
||||
const baseCost = toml.cost;
|
||||
if (baseCost !== undefined && baseCost !== null && typeof baseCost === "object" && !Array.isArray(baseCost)) {
|
||||
if (cost === undefined) {
|
||||
omit.push("cost");
|
||||
} else {
|
||||
for (const key of ["reasoning", "cache_read", "cache_write", "input_audio", "output_audio", "tiers"] as const) {
|
||||
if ((baseCost as Record<string, unknown>)[key] !== undefined && cost[key] === undefined) {
|
||||
omit.push(`cost.${key}`);
|
||||
}
|
||||
}
|
||||
if (hasLegacyContextOver200k(baseCost) && cost.tiers === undefined) {
|
||||
omit.push("cost.context_over_200k");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const baseLimit = toml.limit;
|
||||
if (
|
||||
baseLimit !== undefined &&
|
||||
baseLimit !== null &&
|
||||
typeof baseLimit === "object" &&
|
||||
!Array.isArray(baseLimit) &&
|
||||
(baseLimit as Record<string, unknown>).input !== undefined &&
|
||||
limit.input === undefined
|
||||
) {
|
||||
omit.push("limit.input");
|
||||
}
|
||||
|
||||
return omit.length > 0 ? omit : undefined;
|
||||
}
|
||||
|
||||
function hasLegacyContextOver200k(cost: object) {
|
||||
const tiers = (cost as { tiers?: unknown }).tiers;
|
||||
if (!Array.isArray(tiers) || tiers.length !== 1) return false;
|
||||
|
||||
const tier = tiers[0];
|
||||
if (tier === null || typeof tier !== "object" || Array.isArray(tier)) return false;
|
||||
const tierConfig = (tier as { tier?: unknown }).tier;
|
||||
if (tierConfig === null || typeof tierConfig !== "object" || Array.isArray(tierConfig)) return false;
|
||||
|
||||
const size = (tierConfig as { size?: unknown }).size;
|
||||
return typeof size === "number" && size >= 200_000;
|
||||
}
|
||||
|
||||
function canonicalRuntimeOverrides(
|
||||
provider: string,
|
||||
modelID: string,
|
||||
values: Pick<SyncedFullModel, "name" | "attachment" | "reasoning">,
|
||||
) {
|
||||
const toml = canonicalToml(provider, modelID);
|
||||
return Object.fromEntries(
|
||||
Object.entries(values).filter(([key, value]) => value !== undefined && toml[key] !== value),
|
||||
);
|
||||
}
|
||||
|
||||
function canonicalToml(provider: string, modelID: string) {
|
||||
const key = `${provider}/${modelID}`;
|
||||
let toml = canonicalTomlByModel.get(key);
|
||||
if (toml === undefined) {
|
||||
const filePath = path.join(PROVIDERS_DIR, provider, "models", `${modelID}.toml`);
|
||||
toml = Bun.TOML.parse(readFileSync(filePath, "utf8")) as Record<string, unknown>;
|
||||
canonicalTomlByModel.set(key, toml);
|
||||
}
|
||||
return toml;
|
||||
}
|
||||
|
||||
function canonicalCandidates(provider: string, modelID: string) {
|
||||
const candidates = [modelID];
|
||||
|
||||
if (provider === "anthropic") {
|
||||
candidates.push(modelID.replace(/(claude-(?:opus|sonnet|haiku)-\d+)\.(\d+)/, "$1-$2"));
|
||||
candidates.push(modelID.replace(/^claude-3\.5-/, "claude-3-5-"));
|
||||
}
|
||||
|
||||
if (provider === "llama") {
|
||||
candidates.push(modelID.replace(/^llama-(\d+)-(\d+)/, "llama-$1.$2"));
|
||||
candidates.push(modelID.replace(/^llama-(4)-(maverick|scout)$/, "llama-$1-$2-17b"));
|
||||
}
|
||||
|
||||
if (provider === "mistral") {
|
||||
candidates.push(modelID.replace(/-latest$/, ""));
|
||||
}
|
||||
|
||||
if (provider === "minimax") {
|
||||
candidates.push(modelID.replace(/^minimax-m/, "MiniMax-M"));
|
||||
}
|
||||
|
||||
return [...new Set(candidates)];
|
||||
}
|
||||
@@ -0,0 +1,211 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js";
|
||||
|
||||
const API_BASE = "https://api.x.ai/v1";
|
||||
|
||||
const XAIModel = z.object({
|
||||
id: z.string(),
|
||||
canonical_id: z.string().optional(),
|
||||
created: z.number().int().nonnegative(),
|
||||
aliases: z.array(z.string()).optional(),
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
prompt_text_token_price: z.number().int().nonnegative().optional(),
|
||||
cached_prompt_text_token_price: z.number().int().nonnegative().optional(),
|
||||
completion_text_token_price: z.number().int().nonnegative().optional(),
|
||||
max_prompt_length: z.number().int().nonnegative().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const XAIModelList = z.object({
|
||||
models: z.array(XAIModel),
|
||||
}).passthrough();
|
||||
|
||||
const XAIResponse = z.object({
|
||||
models: z.array(XAIModel),
|
||||
});
|
||||
|
||||
const XAIAPIKey = z.object({
|
||||
acls: z.array(z.string()),
|
||||
}).passthrough();
|
||||
|
||||
type XAIModel = z.infer<typeof XAIModel>;
|
||||
|
||||
export const xai = {
|
||||
id: "xai",
|
||||
name: "xAI",
|
||||
modelsDir: "providers/xai/models",
|
||||
skipCreates: true,
|
||||
sourceID(model) {
|
||||
return model.id;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} xAI models returned by the API were not created because the Models API does not provide enough authoritative metadata for the catalog, especially output token limits and some feature/capability flags. Existing models are still updated from API-authoritative fields.`,
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
async fetchModels() {
|
||||
const key = process.env.XAI_API_KEY;
|
||||
if (key === undefined) throw new Error("xAI sync requires XAI_API_KEY");
|
||||
await assertFullModelAccess(key);
|
||||
|
||||
const models = await Promise.all([
|
||||
fetchTypedModels(key, "language-models"),
|
||||
fetchTypedModels(key, "image-generation-models"),
|
||||
fetchTypedModels(key, "video-generation-models"),
|
||||
]);
|
||||
|
||||
return { models: models.flat() };
|
||||
},
|
||||
parseModels(raw) {
|
||||
const models = XAIResponse.parse(raw).models;
|
||||
const seen = new Set<string>();
|
||||
const expanded: XAIModel[] = [];
|
||||
|
||||
for (const model of models) {
|
||||
if (!seen.has(model.id)) {
|
||||
seen.add(model.id);
|
||||
expanded.push(model);
|
||||
}
|
||||
}
|
||||
|
||||
for (const model of models) {
|
||||
for (const alias of model.aliases ?? []) {
|
||||
if (seen.has(alias)) continue;
|
||||
seen.add(alias);
|
||||
expanded.push({ ...model, id: alias, canonical_id: model.id });
|
||||
}
|
||||
}
|
||||
|
||||
return expanded;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const existing = context.existing(model.id);
|
||||
if (existing === undefined) return undefined;
|
||||
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildModel(model, existing),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<XAIModel>;
|
||||
|
||||
async function assertFullModelAccess(key: string) {
|
||||
const response = await fetch(`${API_BASE}/api-key`, {
|
||||
headers: { Authorization: `Bearer ${key}` },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`xAI API key metadata request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const apiKey = XAIAPIKey.parse(await response.json());
|
||||
if (!apiKey.acls.includes("api-key:model:*")) {
|
||||
throw new Error("xAI sync requires XAI_API_KEY to include api-key:model:* so the model list is not ACL-filtered");
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchTypedModels(key: string, endpoint: string) {
|
||||
const response = await fetch(`${API_BASE}/${endpoint}`, {
|
||||
headers: { Authorization: `Bearer ${key}` },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`xAI ${endpoint} request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
return XAIModelList.parse(await response.json()).models;
|
||||
}
|
||||
|
||||
function dateFromTimestamp(timestamp: number) {
|
||||
return new Date(timestamp * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[] | undefined, fallback: Modality[]) {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = (values ?? [])
|
||||
.map((value) => value.toLowerCase())
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
if (result.includes("image")) result.push("pdf");
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
|
||||
function tokenPrice(value: number | undefined) {
|
||||
if (value === undefined) return undefined;
|
||||
return value / 10_000;
|
||||
}
|
||||
|
||||
function preservedCostTiers(existing: ExistingModel) {
|
||||
// The xAI models API exposes base pricing only; long-context tiers are curated from xAI docs/console.
|
||||
return existing.cost?.tiers;
|
||||
}
|
||||
|
||||
function cost(model: XAIModel, existing: ExistingModel) {
|
||||
const input = tokenPrice(model.prompt_text_token_price);
|
||||
const output = tokenPrice(model.completion_text_token_price);
|
||||
if (input === undefined || output === undefined) return existing.cost;
|
||||
|
||||
return {
|
||||
input,
|
||||
output,
|
||||
reasoning: existing.cost?.reasoning,
|
||||
cache_read: tokenPrice(model.cached_prompt_text_token_price),
|
||||
cache_write: existing.cost?.cache_write,
|
||||
input_audio: existing.cost?.input_audio,
|
||||
output_audio: existing.cost?.output_audio,
|
||||
tiers: preservedCostTiers(existing),
|
||||
};
|
||||
}
|
||||
|
||||
function buildModel(model: XAIModel, existing: ExistingModel): SyncedModel {
|
||||
const name = existing.name;
|
||||
const attachment = existing.attachment;
|
||||
const reasoning = existing.reasoning;
|
||||
const toolCall = existing.tool_call;
|
||||
const openWeights = existing.open_weights;
|
||||
const limit = existing.limit;
|
||||
const releaseDate = existing.release_date;
|
||||
const lastUpdated = existing.last_updated;
|
||||
|
||||
if (
|
||||
name === undefined
|
||||
|| attachment === undefined
|
||||
|| reasoning === undefined
|
||||
|| toolCall === undefined
|
||||
|| openWeights === undefined
|
||||
|| limit === undefined
|
||||
|| (model.canonical_id !== undefined && releaseDate === undefined)
|
||||
|| (model.canonical_id !== undefined && lastUpdated === undefined)
|
||||
) {
|
||||
throw new Error(`xAI model ${model.id} has incomplete local TOML metadata required for sync`);
|
||||
}
|
||||
|
||||
const input = modalities(model.input_modalities, existing.modalities?.input ?? ["text"]);
|
||||
const output = modalities(model.output_modalities, existing.modalities?.output ?? ["text"]);
|
||||
const created = dateFromTimestamp(model.created);
|
||||
|
||||
return {
|
||||
name,
|
||||
family: existing.family,
|
||||
release_date: model.canonical_id === undefined ? created : releaseDate!,
|
||||
last_updated: model.canonical_id === undefined ? created : lastUpdated!,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning,
|
||||
temperature: existing.temperature,
|
||||
tool_call: toolCall,
|
||||
structured_output: existing.structured_output,
|
||||
knowledge: existing.knowledge,
|
||||
open_weights: openWeights,
|
||||
status: existing.status,
|
||||
interleaved: existing.interleaved,
|
||||
cost: cost(model, existing),
|
||||
limit: {
|
||||
input: limit.input,
|
||||
context: model.max_prompt_length ?? limit.context,
|
||||
output: limit.output,
|
||||
},
|
||||
modalities: { input, output },
|
||||
};
|
||||
}
|
||||
@@ -15,7 +15,8 @@ output = 25.000
|
||||
cache_read = 0.500
|
||||
cache_write = 6.250
|
||||
|
||||
[cost.context_over_200k]
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 10.000
|
||||
output = 37.500
|
||||
cache_read = 1.000
|
||||
|
||||
@@ -16,7 +16,8 @@ output = 180.000
|
||||
cache_read = 0
|
||||
cache_write = 0
|
||||
|
||||
[cost.context_over_200k]
|
||||
[[cost.tiers]]
|
||||
tier = { size = 272_000 }
|
||||
input = 60.000
|
||||
output = 270.000
|
||||
|
||||
|
||||
@@ -16,7 +16,8 @@ output = 15.000
|
||||
cache_read = 0.250
|
||||
cache_write = 0
|
||||
|
||||
[cost.context_over_200k]
|
||||
[[cost.tiers]]
|
||||
tier = { size = 272_000 }
|
||||
input = 5.000
|
||||
output = 22.500
|
||||
|
||||
|
||||
+5
-5
@@ -1,5 +1,5 @@
|
||||
name = "DeepSeek V4 Flash Think"
|
||||
family = "deepseek"
|
||||
name = "DeepSeek V4 Flash (Alibaba Cloud)"
|
||||
family = "deepseek-flash"
|
||||
release_date = "2026-04-24"
|
||||
last_updated = "2026-04-24"
|
||||
attachment = false
|
||||
@@ -14,9 +14,9 @@ open_weights = true
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.154
|
||||
output = 0.308
|
||||
cache_read = 0.0308
|
||||
input = 0.14
|
||||
output = 0.28
|
||||
cache_read = 0.028
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
+13
-11
@@ -1,25 +1,27 @@
|
||||
name = "DeepSeek R1 TEE"
|
||||
name = "DeepSeek V4 Pro (Alibaba Cloud)"
|
||||
family = "deepseek-thinking"
|
||||
release_date = "2025-12-29"
|
||||
last_updated = "2026-01-10"
|
||||
release_date = "2026-04-24"
|
||||
last_updated = "2026-04-24"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-05"
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.30
|
||||
output = 1.20
|
||||
input = 1.69
|
||||
output = 3.38
|
||||
cache_read = 0.13
|
||||
|
||||
[limit]
|
||||
context = 163_840
|
||||
output = 163_840
|
||||
context = 1_000_000
|
||||
output = 384_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
+12
-10
@@ -1,7 +1,7 @@
|
||||
name = "GLM 4.5 TEE"
|
||||
name = "GLM-5.1 (Alibaba Cloud)"
|
||||
family = "glm"
|
||||
release_date = "2025-12-29"
|
||||
last_updated = "2026-01-10"
|
||||
release_date = "2026-03-27"
|
||||
last_updated = "2026-03-27"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -9,17 +9,19 @@ tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.35
|
||||
output = 1.55
|
||||
input = 0.84
|
||||
output = 3.38
|
||||
cache_read = 0.169
|
||||
cache_write = 1.05625
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 65_536
|
||||
context = 200_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
@@ -1,4 +1,4 @@
|
||||
name = "Claude Opus 4.6"
|
||||
name = "Claude Opus 4.6 Thinking"
|
||||
family = "claude-opus"
|
||||
release_date = "2026-02-05"
|
||||
last_updated = "2026-03-13"
|
||||
@@ -6,18 +6,29 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-05-31"
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 5
|
||||
output = 25
|
||||
cache_read = 0.5
|
||||
cache_write = 6.25
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 10
|
||||
output = 37.5
|
||||
cache_read = 1.0
|
||||
cache_write = 12.5
|
||||
|
||||
[limit]
|
||||
context = 200_000
|
||||
output = 32_000
|
||||
context = 1_000_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
|
||||
@@ -6,15 +6,25 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-05-31"
|
||||
open_weights = false
|
||||
|
||||
interleaved = true
|
||||
|
||||
[cost]
|
||||
input = 5
|
||||
output = 25
|
||||
cache_read = 0.5
|
||||
cache_write = 6.25
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 10
|
||||
output = 37.5
|
||||
cache_read = 1.0
|
||||
cache_write = 12.5
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 128_000
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
name = "Claude Opus 4.7 Thinking"
|
||||
family = "claude-opus"
|
||||
release_date = "2026-04-16"
|
||||
last_updated = "2026-04-16"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-01-31"
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 5
|
||||
output = 25
|
||||
cache_read = 0.5
|
||||
cache_write = 6.25
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 10
|
||||
output = 37.5
|
||||
cache_read = 1.0
|
||||
cache_write = 12.5
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -6,15 +6,25 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-01-31"
|
||||
open_weights = false
|
||||
|
||||
interleaved = true
|
||||
|
||||
[cost]
|
||||
input = 5
|
||||
output = 25
|
||||
cache_read = 0.5
|
||||
cache_write = 6.25
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 10
|
||||
output = 37.5
|
||||
cache_read = 1.0
|
||||
cache_write = 12.5
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 128_000
|
||||
|
||||
@@ -1,28 +1,33 @@
|
||||
name = "Claude Sonnet 4.6 Think"
|
||||
name = "Claude Sonnet 4.6 Thinking"
|
||||
family = "claude-sonnet"
|
||||
release_date = "2026-02-17"
|
||||
last_updated = "2026-02-17"
|
||||
last_updated = "2026-03-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-08-31"
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 3.00
|
||||
output = 15.00
|
||||
cache_read = 0.30
|
||||
cache_write = 3.75
|
||||
|
||||
[cost.context_over_200k]
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 6.00
|
||||
output = 22.50
|
||||
cache_read = 0.60
|
||||
cache_write = 7.50
|
||||
|
||||
[limit]
|
||||
context = 200_000
|
||||
context = 1_000_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -1,28 +1,32 @@
|
||||
name = "Claude Sonnet 4.6"
|
||||
family = "claude-sonnet"
|
||||
release_date = "2026-02-17"
|
||||
last_updated = "2026-02-17"
|
||||
last_updated = "2026-03-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-08-31"
|
||||
open_weights = false
|
||||
|
||||
interleaved = true
|
||||
|
||||
[cost]
|
||||
input = 3.00
|
||||
output = 15.00
|
||||
cache_read = 0.30
|
||||
cache_write = 3.75
|
||||
|
||||
[cost.context_over_200k]
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 6.00
|
||||
output = 22.50
|
||||
cache_read = 0.60
|
||||
cache_write = 7.50
|
||||
|
||||
[limit]
|
||||
context = 200_000
|
||||
context = 1_000_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
name = "Coding-GLM-5-Free"
|
||||
family = "glm"
|
||||
release_date = "2026-02-11"
|
||||
last_updated = "2026-02-11"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.0
|
||||
output = 0.0
|
||||
|
||||
[limit]
|
||||
context = 204800
|
||||
output = 131072
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
+11
-11
@@ -1,7 +1,7 @@
|
||||
name = "Qwen3 235B A22B"
|
||||
family = "qwen"
|
||||
release_date = "2025-12-29"
|
||||
last_updated = "2026-01-10"
|
||||
name = "Coding GLM 5.1 (free)"
|
||||
family = "glm-free"
|
||||
release_date = "2026-04-11"
|
||||
last_updated = "2026-04-11"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -9,17 +9,17 @@ tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.30
|
||||
output = 1.20
|
||||
input = 0
|
||||
output = 0
|
||||
|
||||
[limit]
|
||||
context = 40_960
|
||||
output = 40_960
|
||||
context = 200_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
@@ -1,4 +1,4 @@
|
||||
name = "Coding-GLM-5.1"
|
||||
name = "Coding GLM 5.1"
|
||||
family = "glm"
|
||||
release_date = "2026-04-11"
|
||||
last_updated = "2026-04-11"
|
||||
@@ -6,7 +6,8 @@ attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
@@ -14,10 +15,11 @@ field = "reasoning_content"
|
||||
[cost]
|
||||
input = 0.06
|
||||
output = 0.22
|
||||
cache_read = 0.013
|
||||
|
||||
[limit]
|
||||
context = 200000
|
||||
output = 128000
|
||||
context = 200_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
|
||||
@@ -1,20 +1,24 @@
|
||||
name = "Coding-MiniMax-M2.7-Free"
|
||||
family = "minimax"
|
||||
name = "Coding MiniMax M2.7 (Free)"
|
||||
family = "minimax-free"
|
||||
release_date = "2026-03-18"
|
||||
last_updated = "2026-03-18"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0
|
||||
output = 0
|
||||
|
||||
[limit]
|
||||
context = 204_800
|
||||
output = 13_100
|
||||
output = 128_100
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
|
||||
+11
-11
@@ -1,7 +1,7 @@
|
||||
name = "Hermes 4 70B"
|
||||
family = "nousresearch"
|
||||
release_date = "2025-12-29"
|
||||
last_updated = "2026-01-10"
|
||||
name = "Coding MiniMax M2.7 Highspeed"
|
||||
family = "minimax"
|
||||
release_date = "2026-03-18"
|
||||
last_updated = "2026-03-18"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -9,17 +9,17 @@ tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.11
|
||||
output = 0.38
|
||||
input = 0.2
|
||||
output = 0.2
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 131_072
|
||||
context = 204_800
|
||||
output = 128_100
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
+10
-10
@@ -1,7 +1,7 @@
|
||||
name = "MiniMax M2.1 TEE"
|
||||
name = "Coding MiniMax M2.7"
|
||||
family = "minimax"
|
||||
release_date = "2025-12-29"
|
||||
last_updated = "2026-01-27"
|
||||
release_date = "2026-03-18"
|
||||
last_updated = "2026-03-18"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -9,17 +9,17 @@ tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.27
|
||||
output = 1.12
|
||||
input = 0.2
|
||||
output = 0.2
|
||||
|
||||
[limit]
|
||||
context = 196_608
|
||||
output = 65_536
|
||||
context = 204_800
|
||||
output = 128_100
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
@@ -0,0 +1,17 @@
|
||||
name = "Coding Xiaomi MiMo-V2.5-Pro"
|
||||
family = "mimo-v2.5-pro"
|
||||
last_updated = "2026-05-13"
|
||||
|
||||
[extends]
|
||||
from = "xiaomi/mimo-v2.5-pro"
|
||||
|
||||
[cost]
|
||||
input = 0.20
|
||||
output = 0.60
|
||||
cache_read = 0.04
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 0.40
|
||||
output = 1.20
|
||||
cache_read = 0.08
|
||||
@@ -0,0 +1,17 @@
|
||||
name = "Coding Xiaomi MiMo-V2.5"
|
||||
family = "mimo-v2.5"
|
||||
last_updated = "2026-05-13"
|
||||
|
||||
[extends]
|
||||
from = "xiaomi/mimo-v2.5"
|
||||
|
||||
[cost]
|
||||
input = 0.08
|
||||
output = 0.40
|
||||
cache_read = 0.016
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 0.16
|
||||
output = 0.80
|
||||
cache_read = 0.032
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
name = "DeepSeek V4 Flash"
|
||||
name = "DeepSeek V4 Flash (DeepSeek)"
|
||||
family = "deepseek-flash"
|
||||
release_date = "2026-04-24"
|
||||
last_updated = "2026-04-24"
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
name = "DeepSeek V4 Pro"
|
||||
name = "DeepSeek V4 Pro (DeepSeek)"
|
||||
family = "deepseek-thinking"
|
||||
release_date = "2026-04-24"
|
||||
last_updated = "2026-04-24"
|
||||
@@ -0,0 +1,38 @@
|
||||
name = "Doubao Seed 2.0 Code Preview"
|
||||
family = "seed"
|
||||
release_date = "2026-02-14"
|
||||
last_updated = "2026-02-14"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.48
|
||||
output = 2.41
|
||||
cache_read = 0.09644
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 32_000 }
|
||||
input = 0.72
|
||||
output = 3.62
|
||||
cache_read = 0.144656
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 128_000 }
|
||||
input = 1.45
|
||||
output = 7.23
|
||||
cache_read = 0.28932
|
||||
|
||||
[limit]
|
||||
context = 256_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,41 @@
|
||||
name = "Doubao Seed 2.0 Lite 260428"
|
||||
family = "seed"
|
||||
release_date = "2026-04-28"
|
||||
last_updated = "2026-04-28"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.08
|
||||
output = 0.51
|
||||
cache_read = 0.01692
|
||||
input_audio = 1.269
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 32_000 }
|
||||
input = 0.13
|
||||
output = 0.76
|
||||
cache_read = 0.02536
|
||||
input_audio = 1.902
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 128_000 }
|
||||
input = 0.25
|
||||
output = 1.52
|
||||
cache_read = 0.05072
|
||||
input_audio = 3.804
|
||||
|
||||
[limit]
|
||||
context = 256_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,41 @@
|
||||
name = "Doubao Seed 2.0 Mini 260428"
|
||||
family = "seed"
|
||||
release_date = "2026-04-28"
|
||||
last_updated = "2026-04-28"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.03
|
||||
output = 0.28
|
||||
cache_read = 0.00564
|
||||
input_audio = 0.423
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 32_000 }
|
||||
input = 0.06
|
||||
output = 0.56
|
||||
cache_read = 0.01128
|
||||
input_audio = 0.846
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 128_000 }
|
||||
input = 0.11
|
||||
output = 1.13
|
||||
cache_read = 0.02256
|
||||
input_audio = 1.692
|
||||
|
||||
[limit]
|
||||
context = 256_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,38 @@
|
||||
name = "Doubao Seed 2.0 Pro"
|
||||
family = "seed"
|
||||
release_date = "2026-02-14"
|
||||
last_updated = "2026-02-14"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.48
|
||||
output = 2.41
|
||||
cache_read = 0.09644
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 32_000 }
|
||||
input = 0.72
|
||||
output = 3.62
|
||||
cache_read = 0.144656
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 128_000 }
|
||||
input = 1.45
|
||||
output = 7.23
|
||||
cache_read = 0.28932
|
||||
|
||||
[limit]
|
||||
context = 256_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -12,8 +12,9 @@ open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.3
|
||||
output = 2.499
|
||||
output = 2.50
|
||||
cache_read = 0.03
|
||||
input_audio = 1.00
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
|
||||
@@ -15,6 +15,12 @@ input = 1.25
|
||||
output = 10
|
||||
cache_read = 0.125
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 2.50
|
||||
output = 15.00
|
||||
cache_read = 0.25
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
@@ -15,6 +15,12 @@ input = 0.5
|
||||
output = 3
|
||||
cache_read = 0.05
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 0.50
|
||||
output = 3.00
|
||||
cache_read = 0.05
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
+5
-4
@@ -1,7 +1,7 @@
|
||||
name = "Gemini 3.1 Flash Lite Preview"
|
||||
name = "Gemini 3.1 Flash Lite"
|
||||
family = "gemini-flash-lite"
|
||||
release_date = "2026-03-03"
|
||||
last_updated = "2026-03-03"
|
||||
release_date = "2026-05-07"
|
||||
last_updated = "2026-05-07"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -13,7 +13,8 @@ open_weights = false
|
||||
[cost]
|
||||
input = 0.25
|
||||
output = 1.5
|
||||
cache_read = 0.25
|
||||
cache_read = 0.025
|
||||
cache_write = 1.00
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
+13
-7
@@ -1,19 +1,25 @@
|
||||
name = "Gemini 2.5 Pro Preview 05-06"
|
||||
name = "Gemini 3.1 Pro Preview Custom Tools"
|
||||
family = "gemini-pro"
|
||||
release_date = "2025-05-06"
|
||||
last_updated = "2025-05-06"
|
||||
release_date = "2026-02-19"
|
||||
last_updated = "2026-02-19"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
knowledge = "2025-01"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 1.25
|
||||
output = 10.00
|
||||
cache_read = 0.31
|
||||
input = 2
|
||||
output = 12
|
||||
cache_read = 0.2
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 4
|
||||
output = 18
|
||||
cache_read = 0.4
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
@@ -15,6 +15,12 @@ input = 2
|
||||
output = 12
|
||||
cache_read = 0.2
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 4.00
|
||||
output = 18.00
|
||||
cache_read = 0.40
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
@@ -1,25 +0,0 @@
|
||||
name = "GLM-5"
|
||||
family = "glm"
|
||||
release_date = "2026-02-11"
|
||||
last_updated = "2026-02-11"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.88
|
||||
output = 2.816
|
||||
cache_read = 0.176
|
||||
|
||||
[limit]
|
||||
context = 202_752
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
+9
-9
@@ -1,11 +1,10 @@
|
||||
name = "Qwen3.6-27B"
|
||||
family = "qwen3.6"
|
||||
release_date = "2026-04-02"
|
||||
last_updated = "2026-04-02"
|
||||
name = "GLM 5 Vision Turbo"
|
||||
family = "glmv"
|
||||
release_date = "2026-05-09"
|
||||
last_updated = "2026-05-09"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
knowledge = "2025-04"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
@@ -14,12 +13,13 @@ open_weights = false
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.6
|
||||
output = 3.6
|
||||
input = 0.7042
|
||||
output = 3.09848
|
||||
cache_read = 0.169008
|
||||
|
||||
[limit]
|
||||
context = 262_144
|
||||
output = 65_536
|
||||
context = 200_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
@@ -1,23 +0,0 @@
|
||||
name = "GPT-4.1 mini"
|
||||
family = "gpt-mini"
|
||||
release_date = "2025-04-14"
|
||||
last_updated = "2025-04-14"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-04"
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.40
|
||||
output = 1.60
|
||||
cache_read = 0.10
|
||||
|
||||
[limit]
|
||||
context = 1_047_576
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
@@ -1,23 +0,0 @@
|
||||
name = "GPT-4.1"
|
||||
family = "gpt"
|
||||
release_date = "2025-04-14"
|
||||
last_updated = "2025-04-14"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-04"
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 2.00
|
||||
output = 8.00
|
||||
cache_read = 0.50
|
||||
|
||||
[limit]
|
||||
context = 1_047_576
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
@@ -5,20 +5,20 @@ last_updated = "2025-11-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2024-09-30"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2024-09-30"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.25
|
||||
output = 2
|
||||
output = 2.00
|
||||
cache_read = 0.025
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
output = 128_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
|
||||
@@ -5,20 +5,20 @@ last_updated = "2025-11-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2024-09-30"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2024-09-30"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 1.25
|
||||
output = 10
|
||||
output = 10.00
|
||||
cache_read = 0.125
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
output = 128_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
|
||||
@@ -1,21 +1,23 @@
|
||||
name = "GPT-5.1"
|
||||
family = "gpt"
|
||||
release_date = "2025-11-15"
|
||||
last_updated = "2025-11-15"
|
||||
release_date = "2025-11-13"
|
||||
last_updated = "2025-11-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
temperature = false
|
||||
knowledge = "2024-09-30"
|
||||
tool_call = true
|
||||
knowledge = "2025-11"
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 1.25
|
||||
output = 10.00
|
||||
cache_read = 0.125
|
||||
cache_read = 0.13
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
name = "GPT-5.2-Codex"
|
||||
name = "GPT-5.2 Codex"
|
||||
family = "gpt-codex"
|
||||
release_date = "2026-01-14"
|
||||
last_updated = "2026-01-14"
|
||||
release_date = "2025-12-11"
|
||||
last_updated = "2025-12-11"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2025-08-31"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
@@ -16,8 +17,9 @@ cache_read = 0.175
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
@@ -7,6 +7,7 @@ reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2025-08-31"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
@@ -16,6 +17,7 @@ cache_read = 0.175
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -5,20 +5,20 @@ last_updated = "2026-02-05"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2025-08-31"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-08-31"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 1.75
|
||||
output = 14
|
||||
output = 14.00
|
||||
cache_read = 0.175
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
output = 128_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
|
||||
@@ -1,13 +1,14 @@
|
||||
name = "GPT-5.4-Mini"
|
||||
name = "GPT-5.4 mini"
|
||||
family = "gpt-mini"
|
||||
release_date = "2026-03-11"
|
||||
last_updated = "2026-03-11"
|
||||
release_date = "2026-03-17"
|
||||
last_updated = "2026-03-17"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2025-08-31"
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.75
|
||||
@@ -16,8 +17,13 @@ cache_read = 0.075
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
input = 272_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
|
||||
[experimental.modes.fast]
|
||||
cost = { input = 1.50, output = 9.00, cache_read = 0.15 }
|
||||
provider = { body = { service_tier = "priority" } }
|
||||
|
||||
@@ -1,23 +1,35 @@
|
||||
name = "GPT-5.4"
|
||||
family = "gpt"
|
||||
release_date = "2026-03-11"
|
||||
last_updated = "2026-03-11"
|
||||
release_date = "2026-03-05"
|
||||
last_updated = "2026-03-05"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2025-08-31"
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 2.50
|
||||
output = 15.00
|
||||
cache_read = 0.25
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 272_000 }
|
||||
input = 5.00
|
||||
output = 22.50
|
||||
cache_read = 0.50
|
||||
|
||||
[limit]
|
||||
context = 400_000
|
||||
context = 1_050_000
|
||||
input = 922_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[experimental.modes.fast]
|
||||
cost = { input = 5.00, output = 30.00, cache_read = 0.50 }
|
||||
provider = { body = { service_tier = "priority" } }
|
||||
|
||||
@@ -5,20 +5,31 @@ last_updated = "2026-04-23"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
knowledge = "2025-12-01"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-12-01"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 5
|
||||
output = 30
|
||||
cache_read = 0.5
|
||||
input = 5.00
|
||||
output = 30.00
|
||||
cache_read = 0.50
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 272_000 }
|
||||
input = 10.00
|
||||
output = 45.00
|
||||
cache_read = 1.00
|
||||
|
||||
[limit]
|
||||
context = 1_050_000
|
||||
input = 922_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[experimental.modes.fast]
|
||||
cost = { input = 12.50, output = 75.00, cache_read = 1.25 }
|
||||
provider = { body = { service_tier = "priority" } }
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
name = "Grok 4.3"
|
||||
family = "grok"
|
||||
release_date = "2026-05-01"
|
||||
last_updated = "2026-05-01"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 1.25
|
||||
output = 2.5
|
||||
cache_read = 0.2
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 200_000 }
|
||||
input = 2.5
|
||||
output = 5.0
|
||||
cache_read = 0.4
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 1_000_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
@@ -1,23 +0,0 @@
|
||||
name = "Kimi-K2-Thinking"
|
||||
family = "kimi"
|
||||
release_date = "2025-11-06"
|
||||
last_updated = "2025-11-06"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
knowledge = "2025-11"
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
[cost]
|
||||
input = 0.55
|
||||
output = 2.19
|
||||
cache_read = 0.14
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -2,7 +2,7 @@ name = "Kimi K2.5"
|
||||
family = "kimi-k2.5"
|
||||
release_date = "2026-01"
|
||||
last_updated = "2026-01"
|
||||
attachment = false
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
@@ -16,11 +16,11 @@ field = "reasoning_content"
|
||||
[cost]
|
||||
input = 0.6
|
||||
output = 3
|
||||
cache_read = 0.105
|
||||
cache_read = 0.10
|
||||
|
||||
[limit]
|
||||
context = 256_000
|
||||
output = 0
|
||||
context = 262_144
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
|
||||
@@ -4,7 +4,7 @@ release_date = "2026-04-21"
|
||||
last_updated = "2026-04-21"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-01"
|
||||
@@ -15,12 +15,12 @@ field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.95
|
||||
output = 3.9995
|
||||
cache_read = 0.160835
|
||||
output = 4
|
||||
cache_read = 0.16
|
||||
|
||||
[limit]
|
||||
context = 262_144
|
||||
output = 262_144
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
name = "MiniMax-M2.1"
|
||||
family = "minimax"
|
||||
release_date = "2025-12-23"
|
||||
last_updated = "2025-12-23"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_details"
|
||||
|
||||
[cost]
|
||||
input = 0.288
|
||||
output = 1.152
|
||||
|
||||
[limit]
|
||||
context = 204_800
|
||||
output = 192_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -1,4 +1,4 @@
|
||||
name = "MiniMax-M2.7"
|
||||
name = "MiniMax M2.7"
|
||||
family = "minimax"
|
||||
release_date = "2026-03-18"
|
||||
last_updated = "2026-03-18"
|
||||
@@ -6,15 +6,20 @@ attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.2958
|
||||
output = 1.1832
|
||||
cache_read = 0.05916
|
||||
input = 0.3
|
||||
output = 1.2
|
||||
cache_read = 0.06
|
||||
cache_write = 0.375
|
||||
|
||||
[limit]
|
||||
context = 200_000
|
||||
context = 204_800
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
name = "Qwen3 Coder Plus"
|
||||
family = "qwen"
|
||||
release_date = "2025-07-23"
|
||||
last_updated = "2025-07-23"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = true
|
||||
knowledge = "2025-04"
|
||||
open_weights = true
|
||||
|
||||
[cost]
|
||||
input = 0.137
|
||||
output = 0.548
|
||||
cache_read = 0.137
|
||||
|
||||
[limit]
|
||||
context = 2_000_000
|
||||
output = 64_000
|
||||
input = 262_144
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -1,23 +0,0 @@
|
||||
name = "Qwen3 Max"
|
||||
family = "qwen"
|
||||
release_date = "2025-09-23"
|
||||
last_updated = "2025-09-23"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = true
|
||||
knowledge = "2025-04"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.34246
|
||||
output = 1.36984
|
||||
cache_read = 0.34246
|
||||
|
||||
[limit]
|
||||
context = 252_000
|
||||
output = 32_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -1,24 +0,0 @@
|
||||
name = "Qwen3.5 Plus"
|
||||
family = "qwen"
|
||||
release_date = "2026-02-16"
|
||||
last_updated = "2026-02-16"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
knowledge = "2025-04"
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.1096
|
||||
output = 0.6576
|
||||
cache_read = 0.01096
|
||||
cache_write = 0.137
|
||||
|
||||
[limit]
|
||||
context = 991_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -1,23 +1,34 @@
|
||||
name = "Qwen3.6 Plus"
|
||||
family = "qwen"
|
||||
name = "Qwen3.6 Flash"
|
||||
family = "qwen3.6"
|
||||
release_date = "2026-04-02"
|
||||
last_updated = "2026-04-02"
|
||||
attachment = false
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-04"
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.169
|
||||
output = 1.014
|
||||
input = 0.17
|
||||
output = 1.01
|
||||
cache_read = 0.0169
|
||||
cache_write = 0.21125
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 0.68
|
||||
output = 4.06
|
||||
cache_read = 0.0676
|
||||
cache_write = 0.845
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 65_536
|
||||
context = 991_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
name = "Qwen3.6 Max Preview"
|
||||
family = "qwen3.6"
|
||||
release_date = "2026-05-09"
|
||||
last_updated = "2026-05-09"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-04"
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 1.27
|
||||
output = 7.61
|
||||
cache_read = 0.1268
|
||||
cache_write = 1.585
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 128_000 }
|
||||
input = 2.11
|
||||
output = 12.67
|
||||
cache_read = 0.2112
|
||||
cache_write = 2.64
|
||||
|
||||
[limit]
|
||||
context = 240_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,35 @@
|
||||
name = "Qwen3.6 Plus"
|
||||
family = "qwen3.6"
|
||||
release_date = "2026-05-09"
|
||||
last_updated = "2026-05-09"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-04"
|
||||
open_weights = false
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
|
||||
[cost]
|
||||
input = 0.28
|
||||
output = 1.69
|
||||
cache_read = 0.0282
|
||||
cache_write = 0.3525
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 1.13
|
||||
output = 6.77
|
||||
cache_read = 0.1128
|
||||
cache_write = 1.41
|
||||
|
||||
[limit]
|
||||
context = 991_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,16 @@
|
||||
name = "Xiaomi MiMo-V2.5 (free)"
|
||||
family = "mimo-v2.5"
|
||||
last_updated = "2026-05-13"
|
||||
|
||||
[extends]
|
||||
from = "xiaomi/mimo-v2.5"
|
||||
omit = ["cost.context_over_200k", "cost.tiers"]
|
||||
|
||||
[cost]
|
||||
input = 0
|
||||
output = 0
|
||||
cache_read = 0
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 131_072
|
||||
@@ -0,0 +1,16 @@
|
||||
name = "Xiaomi MiMo-V2.5-Pro (free)"
|
||||
family = "mimo-v2.5-pro"
|
||||
last_updated = "2026-05-13"
|
||||
|
||||
[extends]
|
||||
from = "xiaomi/mimo-v2.5-pro"
|
||||
omit = ["cost.context_over_200k", "cost.tiers"]
|
||||
|
||||
[cost]
|
||||
input = 0
|
||||
output = 0
|
||||
cache_read = 0
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 131_072
|
||||
@@ -0,0 +1,17 @@
|
||||
name = "Xiaomi MiMo-V2.5-Pro"
|
||||
family = "mimo-v2.5-pro"
|
||||
last_updated = "2026-05-13"
|
||||
|
||||
[extends]
|
||||
from = "xiaomi/mimo-v2.5-pro"
|
||||
|
||||
[cost]
|
||||
input = 1.10
|
||||
output = 3.30
|
||||
cache_read = 0.22
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 2.20
|
||||
output = 6.60
|
||||
cache_read = 0.44
|
||||
@@ -0,0 +1,17 @@
|
||||
name = "Xiaomi MiMo-V2.5"
|
||||
family = "mimo-v2.5"
|
||||
last_updated = "2026-05-13"
|
||||
|
||||
[extends]
|
||||
from = "xiaomi/mimo-v2.5"
|
||||
|
||||
[cost]
|
||||
input = 0.44
|
||||
output = 2.20
|
||||
cache_read = 0.088
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 0.88
|
||||
output = 4.40
|
||||
cache_read = 0.176
|
||||
@@ -1,4 +1,4 @@
|
||||
name = "GLM-5.1"
|
||||
name = "GLM-5.1 (Z.ai)"
|
||||
family = "glm"
|
||||
release_date = "2026-03-27"
|
||||
last_updated = "2026-03-27"
|
||||
@@ -7,7 +7,7 @@ reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
open_weights = true
|
||||
|
||||
[interleaved]
|
||||
field = "reasoning_content"
|
||||
@@ -10,10 +10,17 @@ tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.276
|
||||
output = 1.651
|
||||
cache_read = 0.028
|
||||
cache_write = 0.344
|
||||
input = 0.50
|
||||
output = 3.00
|
||||
cache_read = 0.05
|
||||
cache_write = 0.625
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 2.00
|
||||
output = 6.00
|
||||
cache_read = 0.20
|
||||
cache_write = 2.50
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
|
||||
@@ -10,10 +10,17 @@ tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 0.276
|
||||
output = 1.651
|
||||
cache_read = 0.028
|
||||
cache_write = 0.344
|
||||
input = 0.50
|
||||
output = 3.00
|
||||
cache_read = 0.05
|
||||
cache_write = 0.625
|
||||
|
||||
[[cost.tiers]]
|
||||
tier = { size = 256_000 }
|
||||
input = 2.00
|
||||
output = 6.00
|
||||
cache_read = 0.20
|
||||
cache_write = 2.50
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
name = "Qwen3.7 Max"
|
||||
family = "qwen"
|
||||
release_date = "2026-05-21"
|
||||
last_updated = "2026-05-21"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 2.50
|
||||
output = 7.50
|
||||
cache_read = 0.50
|
||||
cache_write = 3.125
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -1,2 +0,0 @@
|
||||
[extends]
|
||||
from = "anthropic/claude-3-5-haiku-20241022"
|
||||
@@ -1,2 +0,0 @@
|
||||
[extends]
|
||||
from = "anthropic/claude-3-5-sonnet-20240620"
|
||||
@@ -1,24 +0,0 @@
|
||||
name = "Claude Sonnet 3.5 v2"
|
||||
family = "claude-sonnet"
|
||||
release_date = "2024-10-22"
|
||||
last_updated = "2024-10-22"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-04"
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 3.00
|
||||
output = 15.00
|
||||
cache_read = 0.30
|
||||
cache_write = 3.75
|
||||
|
||||
[limit]
|
||||
context = 200_000
|
||||
output = 8_192
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -1,24 +0,0 @@
|
||||
name = "Claude Sonnet 3.7"
|
||||
family = "claude-sonnet"
|
||||
release_date = "2025-02-19"
|
||||
last_updated = "2025-02-19"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-04"
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[cost]
|
||||
input = 3.00
|
||||
output = 15.00
|
||||
cache_read = 0.30
|
||||
cache_write = 3.75
|
||||
|
||||
[limit]
|
||||
context = 200_000
|
||||
output = 8_192
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -1,2 +0,0 @@
|
||||
[extends]
|
||||
from = "anthropic/claude-opus-4-20250514"
|
||||
@@ -6,7 +6,6 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-01-31"
|
||||
open_weights = false
|
||||
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
[extends]
|
||||
from = "anthropic/claude-sonnet-4-20250514"
|
||||
@@ -1,2 +1,4 @@
|
||||
structured_output = true
|
||||
|
||||
[extends]
|
||||
from = "anthropic/claude-sonnet-4-6"
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
name = "Claude Haiku 4.5 (AU)"
|
||||
structured_output = true
|
||||
|
||||
[extends]
|
||||
from = "anthropic/claude-haiku-4-5-20251001"
|
||||
@@ -0,0 +1,5 @@
|
||||
name = "Claude Sonnet 4.5 (AU)"
|
||||
structured_output = true
|
||||
|
||||
[extends]
|
||||
from = "anthropic/claude-sonnet-4-5"
|
||||
@@ -7,6 +7,7 @@ reasoning = true
|
||||
temperature = true
|
||||
knowledge = "2024-07"
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[cost]
|
||||
|
||||
@@ -6,7 +6,6 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-01-31"
|
||||
open_weights = false
|
||||
|
||||
|
||||
@@ -1,4 +0,0 @@
|
||||
name = "Claude Sonnet 4 (EU)"
|
||||
|
||||
[extends]
|
||||
from = "anthropic/claude-sonnet-4-20250514"
|
||||
@@ -1,4 +1,5 @@
|
||||
name = "Claude Sonnet 4.6 (EU)"
|
||||
structured_output = true
|
||||
|
||||
[extends]
|
||||
from = "anthropic/claude-sonnet-4-6"
|
||||
|
||||
@@ -6,7 +6,6 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-01-31"
|
||||
open_weights = false
|
||||
|
||||
|
||||
@@ -1,4 +0,0 @@
|
||||
name = "Claude Sonnet 4 (Global)"
|
||||
|
||||
[extends]
|
||||
from = "anthropic/claude-sonnet-4-20250514"
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user