Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 78d7327c0a | |||
| fef032dae3 | |||
| fb890b8d55 |
@@ -1,68 +0,0 @@
|
||||
name: Close stale pull requests
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "17 3 * * *"
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
close-stale-pull-requests:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/github-script@v8
|
||||
env:
|
||||
REVIEWER: rekram1-node
|
||||
with:
|
||||
script: |
|
||||
const { owner, repo } = context.repo
|
||||
const now = Date.now()
|
||||
const weekAgo = now - 7 * 24 * 60 * 60 * 1000
|
||||
const monthAgo = now - 30 * 24 * 60 * 60 * 1000
|
||||
|
||||
const pulls = await github.paginate(github.rest.pulls.list, {
|
||||
owner,
|
||||
repo,
|
||||
state: "open",
|
||||
per_page: 100,
|
||||
})
|
||||
|
||||
const feedbackPulls = new Set()
|
||||
for (const qualifier of ["commenter", "reviewed-by"]) {
|
||||
const results = await github.paginate(
|
||||
github.rest.search.issuesAndPullRequests,
|
||||
{
|
||||
q: `repo:${owner}/${repo} is:pr is:open ${qualifier}:${process.env.REVIEWER}`,
|
||||
per_page: 100,
|
||||
},
|
||||
)
|
||||
|
||||
for (const result of results) feedbackPulls.add(result.number)
|
||||
}
|
||||
|
||||
for (const pull of pulls) {
|
||||
const updatedAt = Date.parse(pull.updated_at)
|
||||
const monthStale = updatedAt < monthAgo
|
||||
const feedbackStale = updatedAt < weekAgo && feedbackPulls.has(pull.number)
|
||||
if (!monthStale && !feedbackStale) continue
|
||||
|
||||
const reason = monthStale
|
||||
? "it has not been updated in 30 days"
|
||||
: `it has not been updated in 7 days after feedback from @${process.env.REVIEWER}`
|
||||
|
||||
await github.rest.issues.createComment({
|
||||
owner,
|
||||
repo,
|
||||
issue_number: pull.number,
|
||||
body: `Closing this pull request as stale because ${reason}. Feel free to reopen it or submit a new pull request if the work is resumed.`,
|
||||
})
|
||||
await github.rest.pulls.update({
|
||||
owner,
|
||||
repo,
|
||||
pull_number: pull.number,
|
||||
state: "closed",
|
||||
})
|
||||
}
|
||||
@@ -20,8 +20,8 @@ jobs:
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Run opencode
|
||||
uses: anomalyco/opencode/github@latest
|
||||
uses: sst/opencode/github@latest
|
||||
env:
|
||||
OPENCODE_API_KEY: ${{ secrets.OPENCODE_API_KEY }}
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
with:
|
||||
model: opencode/gpt-5.5
|
||||
model: anthropic/claude-sonnet-4-20250514
|
||||
|
||||
@@ -63,11 +63,7 @@ jobs:
|
||||
- name: Sync model catalogs
|
||||
run: bun models:sync ${{ matrix.provider }}
|
||||
env:
|
||||
BASETEN_API_KEY: ${{ secrets.BASETEN_API_KEY }}
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
||||
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
|
||||
VENICE_API_KEY: ${{ secrets.VENICE_API_KEY }}
|
||||
LLMGATEWAY_API_KEY: ${{ secrets.LLMGATEWAY_API_KEY }}
|
||||
GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
GOOGLE_GENERATIVE_AI_API_KEY: ${{ secrets.GOOGLE_GENERATIVE_AI_API_KEY }}
|
||||
@@ -85,7 +81,7 @@ jobs:
|
||||
LABELS: automation,model-sync,provider:${{ matrix.provider }}
|
||||
TITLE: "chore(sync): update ${{ matrix.name }} model catalog"
|
||||
run: |
|
||||
if [ -z "$(git status --porcelain -- models providers)" ]; then
|
||||
if [ -z "$(git status --porcelain -- providers)" ]; then
|
||||
echo "No model catalog changes found."
|
||||
exit 0
|
||||
fi
|
||||
@@ -94,7 +90,7 @@ jobs:
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
git fetch --no-tags --depth=1 origin "+refs/heads/$BRANCH:refs/remotes/origin/$BRANCH" || true
|
||||
git checkout -B "$BRANCH"
|
||||
git add models providers
|
||||
git add providers
|
||||
git commit -m "$TITLE"
|
||||
git push --force-with-lease origin "$BRANCH"
|
||||
|
||||
|
||||
+3
-1
@@ -5,4 +5,6 @@ dist
|
||||
.DS_Store
|
||||
.sync/
|
||||
node_modules
|
||||
.opencode/package-lock.json
|
||||
data/tokenspeed-monitor.sqlite
|
||||
data/tokenspeed-monitor.sqlite-shm
|
||||
data/tokenspeed-monitor.sqlite-wal
|
||||
|
||||
Generated
+380
@@ -0,0 +1,380 @@
|
||||
{
|
||||
"name": ".opencode",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"dependencies": {
|
||||
"@opencode-ai/plugin": "1.15.13"
|
||||
}
|
||||
},
|
||||
"node_modules/@msgpackr-extract/msgpackr-extract-darwin-arm64": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-darwin-arm64/-/msgpackr-extract-darwin-arm64-3.0.4.tgz",
|
||||
"integrity": "sha512-LCkGo6JDfaBhgST7UpPWgNgLINpcpabaHfyz5OBx75nUYxBsaEPxjnyNjWpeb/xBup/682QnBfRBy2/LvPutZQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
]
|
||||
},
|
||||
"node_modules/@msgpackr-extract/msgpackr-extract-darwin-x64": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-darwin-x64/-/msgpackr-extract-darwin-x64-3.0.4.tgz",
|
||||
"integrity": "sha512-zExlW9zUJKZH/tOtVMttwjKa4Xm/3KcNjnE3dPN92uCktwavMxpgCA3MoJK/DOnTWsQgo224OaST27/mPNAf+w==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
]
|
||||
},
|
||||
"node_modules/@msgpackr-extract/msgpackr-extract-linux-arm": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-linux-arm/-/msgpackr-extract-linux-arm-3.0.4.tgz",
|
||||
"integrity": "sha512-Tg3yX65f5GbtXLkrYEHE5oibZG9epyYWas7FogTTEJeDEF9JlXJzKgXaNhT3UXlTOeA+AfZpYZYZ0uPj7Cfquw==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
]
|
||||
},
|
||||
"node_modules/@msgpackr-extract/msgpackr-extract-linux-arm64": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-linux-arm64/-/msgpackr-extract-linux-arm64-3.0.4.tgz",
|
||||
"integrity": "sha512-dgX0P/9wGPJeHFBG+ZmhgE6bmtMt7NP5CRBGyyktpopdk/mW4POnrpQsSLtKI1dwpc+pPLuXHDh6vvskyQE/sw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
]
|
||||
},
|
||||
"node_modules/@msgpackr-extract/msgpackr-extract-linux-x64": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-linux-x64/-/msgpackr-extract-linux-x64-3.0.4.tgz",
|
||||
"integrity": "sha512-8TNXMEjJc3QEy7R/x1INhgiU+XakDAFUzBhaz7+Rbrs8NH5UQeHQxxmzsSBJGyV6I1jW79undiQm8tOI+D+8FQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
]
|
||||
},
|
||||
"node_modules/@msgpackr-extract/msgpackr-extract-win32-x64": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-win32-x64/-/msgpackr-extract-win32-x64-3.0.4.tgz",
|
||||
"integrity": "sha512-CmCXPQrkbwExx3j946/PtHWHbYJiCRBRDl4BlkRQcJB/YOwQxJRTpoo7aTsortjgoJ1x7opzTSxn7C+ASSLVjQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"win32"
|
||||
]
|
||||
},
|
||||
"node_modules/@opencode-ai/plugin": {
|
||||
"version": "1.15.13",
|
||||
"resolved": "https://registry.npmjs.org/@opencode-ai/plugin/-/plugin-1.15.13.tgz",
|
||||
"integrity": "sha512-NFwZGhmxIPijtfz9swPJXDmhOpq4UWP8WjEE7GEMr7FwtJrK/hv6v36nFimed5+OKk+pQCrTJn/vhRW7Io72IA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@opencode-ai/sdk": "1.15.13",
|
||||
"effect": "4.0.0-beta.66",
|
||||
"zod": "4.1.8"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"@opentui/core": ">=0.2.16",
|
||||
"@opentui/keymap": ">=0.2.16",
|
||||
"@opentui/solid": ">=0.2.16"
|
||||
},
|
||||
"peerDependenciesMeta": {
|
||||
"@opentui/core": {
|
||||
"optional": true
|
||||
},
|
||||
"@opentui/keymap": {
|
||||
"optional": true
|
||||
},
|
||||
"@opentui/solid": {
|
||||
"optional": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/@opencode-ai/sdk": {
|
||||
"version": "1.15.13",
|
||||
"resolved": "https://registry.npmjs.org/@opencode-ai/sdk/-/sdk-1.15.13.tgz",
|
||||
"integrity": "sha512-4TwojIoQ8EG6/mVBuUVYZXiFcwNmiiytEnjnvyuvSJjGwFIlw2YIBFxtSVC3FbwwbwHT63teh1RHiQUUC4U5xw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"cross-spawn": "7.0.6"
|
||||
}
|
||||
},
|
||||
"node_modules/@standard-schema/spec": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/@standard-schema/spec/-/spec-1.1.0.tgz",
|
||||
"integrity": "sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/cross-spawn": {
|
||||
"version": "7.0.6",
|
||||
"resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz",
|
||||
"integrity": "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"path-key": "^3.1.0",
|
||||
"shebang-command": "^2.0.0",
|
||||
"which": "^2.0.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 8"
|
||||
}
|
||||
},
|
||||
"node_modules/detect-libc": {
|
||||
"version": "2.1.2",
|
||||
"resolved": "https://registry.npmjs.org/detect-libc/-/detect-libc-2.1.2.tgz",
|
||||
"integrity": "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==",
|
||||
"license": "Apache-2.0",
|
||||
"optional": true,
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/effect": {
|
||||
"version": "4.0.0-beta.66",
|
||||
"resolved": "https://registry.npmjs.org/effect/-/effect-4.0.0-beta.66.tgz",
|
||||
"integrity": "sha512-4arEr62cziFa8BBVDUwJCJJmaVepXf/kRg7KtC0h8+bufngscrHbwWFhr9c+HonwOF+31U3iD3xUJmw9KzX7Dw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@standard-schema/spec": "^1.1.0",
|
||||
"fast-check": "^4.6.0",
|
||||
"find-my-way-ts": "^0.1.6",
|
||||
"ini": "^6.0.0",
|
||||
"kubernetes-types": "^1.30.0",
|
||||
"msgpackr": "^1.11.9",
|
||||
"multipasta": "^0.2.7",
|
||||
"toml": "^4.1.1",
|
||||
"uuid": "^13.0.0",
|
||||
"yaml": "^2.8.3"
|
||||
}
|
||||
},
|
||||
"node_modules/fast-check": {
|
||||
"version": "4.8.0",
|
||||
"resolved": "https://registry.npmjs.org/fast-check/-/fast-check-4.8.0.tgz",
|
||||
"integrity": "sha512-GOJ158CUMnN6cSahsv4+ExARvIDuzzinFjkp0E9WtiBa5zcVeLozVkWaE4IzFcc+Y48Wp1EDlUZsXRyAztQcSg==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "individual",
|
||||
"url": "https://github.com/sponsors/dubzzz"
|
||||
},
|
||||
{
|
||||
"type": "opencollective",
|
||||
"url": "https://opencollective.com/fast-check"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"pure-rand": "^8.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=12.17.0"
|
||||
}
|
||||
},
|
||||
"node_modules/find-my-way-ts": {
|
||||
"version": "0.1.6",
|
||||
"resolved": "https://registry.npmjs.org/find-my-way-ts/-/find-my-way-ts-0.1.6.tgz",
|
||||
"integrity": "sha512-a85L9ZoXtNAey3Y6Z+eBWW658kO/MwR7zIafkIUPUMf3isZG0NCs2pjW2wtjxAKuJPxMAsHUIP4ZPGv0o5gyTA==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/ini": {
|
||||
"version": "6.0.0",
|
||||
"resolved": "https://registry.npmjs.org/ini/-/ini-6.0.0.tgz",
|
||||
"integrity": "sha512-IBTdIkzZNOpqm7q3dRqJvMaldXjDHWkEDfrwGEQTs5eaQMWV+djAhR+wahyNNMAa+qpbDUhBMVt4ZKNwpPm7xQ==",
|
||||
"license": "ISC",
|
||||
"engines": {
|
||||
"node": "^20.17.0 || >=22.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/isexe": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/isexe/-/isexe-2.0.0.tgz",
|
||||
"integrity": "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw==",
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/kubernetes-types": {
|
||||
"version": "1.30.0",
|
||||
"resolved": "https://registry.npmjs.org/kubernetes-types/-/kubernetes-types-1.30.0.tgz",
|
||||
"integrity": "sha512-Dew1okvhM/SQcIa2rcgujNndZwU8VnSapDgdxlYoB84ZlpAD43U6KLAFqYo17ykSFGHNPrg0qry0bP+GJd9v7Q==",
|
||||
"license": "Apache-2.0"
|
||||
},
|
||||
"node_modules/msgpackr": {
|
||||
"version": "1.11.12",
|
||||
"resolved": "https://registry.npmjs.org/msgpackr/-/msgpackr-1.11.12.tgz",
|
||||
"integrity": "sha512-RBdJ1Un7yGlXWajrkxcSa93nvQ0w4zBf60c0yYv7YtBelP8H2FA7XsfBbMHtXKXUMUxH7zV3Zuozh+kUQWhHvg==",
|
||||
"license": "MIT",
|
||||
"optionalDependencies": {
|
||||
"msgpackr-extract": "^3.0.2"
|
||||
}
|
||||
},
|
||||
"node_modules/msgpackr-extract": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/msgpackr-extract/-/msgpackr-extract-3.0.4.tgz",
|
||||
"integrity": "sha512-4kmO/MdyUIkLIvTPr8VHLil4AtoKIoniWPIEk5+CDy0xnWC84azhSFmuJ7PxZdsYtiP5kEeQsORAVIeMgxT+Hw==",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"node-gyp-build-optional-packages": "5.2.2"
|
||||
},
|
||||
"bin": {
|
||||
"download-msgpackr-prebuilds": "bin/download-prebuilds.js"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@msgpackr-extract/msgpackr-extract-darwin-arm64": "3.0.4",
|
||||
"@msgpackr-extract/msgpackr-extract-darwin-x64": "3.0.4",
|
||||
"@msgpackr-extract/msgpackr-extract-linux-arm": "3.0.4",
|
||||
"@msgpackr-extract/msgpackr-extract-linux-arm64": "3.0.4",
|
||||
"@msgpackr-extract/msgpackr-extract-linux-x64": "3.0.4",
|
||||
"@msgpackr-extract/msgpackr-extract-win32-x64": "3.0.4"
|
||||
}
|
||||
},
|
||||
"node_modules/multipasta": {
|
||||
"version": "0.2.7",
|
||||
"resolved": "https://registry.npmjs.org/multipasta/-/multipasta-0.2.7.tgz",
|
||||
"integrity": "sha512-KPA58d68KgGil15oDqXjkUBEBYc00XvbPj5/X+dyzeo/lWm9Nc25pQRlf1D+gv4OpK7NM0J1odrbu9JNNGvynA==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/node-gyp-build-optional-packages": {
|
||||
"version": "5.2.2",
|
||||
"resolved": "https://registry.npmjs.org/node-gyp-build-optional-packages/-/node-gyp-build-optional-packages-5.2.2.tgz",
|
||||
"integrity": "sha512-s+w+rBWnpTMwSFbaE0UXsRlg7hU4FjekKU4eyAih5T8nJuNZT1nNsskXpxmeqSK9UzkBl6UgRlnKc8hz8IEqOw==",
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"detect-libc": "^2.0.1"
|
||||
},
|
||||
"bin": {
|
||||
"node-gyp-build-optional-packages": "bin.js",
|
||||
"node-gyp-build-optional-packages-optional": "optional.js",
|
||||
"node-gyp-build-optional-packages-test": "build-test.js"
|
||||
}
|
||||
},
|
||||
"node_modules/path-key": {
|
||||
"version": "3.1.1",
|
||||
"resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz",
|
||||
"integrity": "sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/pure-rand": {
|
||||
"version": "8.4.0",
|
||||
"resolved": "https://registry.npmjs.org/pure-rand/-/pure-rand-8.4.0.tgz",
|
||||
"integrity": "sha512-IoM8YF/jY0hiugFo/wOWqfmarlE6J0wc6fDK1PhftMk7MGhVZl88sZimmqBBFomLOCSmcCCpsfj7wXASCpvK9A==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "individual",
|
||||
"url": "https://github.com/sponsors/dubzzz"
|
||||
},
|
||||
{
|
||||
"type": "opencollective",
|
||||
"url": "https://opencollective.com/fast-check"
|
||||
}
|
||||
],
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/shebang-command": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz",
|
||||
"integrity": "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"shebang-regex": "^3.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/shebang-regex": {
|
||||
"version": "3.0.0",
|
||||
"resolved": "https://registry.npmjs.org/shebang-regex/-/shebang-regex-3.0.0.tgz",
|
||||
"integrity": "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/toml": {
|
||||
"version": "4.1.1",
|
||||
"resolved": "https://registry.npmjs.org/toml/-/toml-4.1.1.tgz",
|
||||
"integrity": "sha512-EBJnVBr3dTXdA89WVFoAIPUqkBjxPMwRqsfuo1r240tKFHXv3zgca4+NJib/h6TyvGF7vOawz0jGuryJCdNHrw==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
}
|
||||
},
|
||||
"node_modules/uuid": {
|
||||
"version": "13.0.2",
|
||||
"resolved": "https://registry.npmjs.org/uuid/-/uuid-13.0.2.tgz",
|
||||
"integrity": "sha512-vzi9uRZ926x4XV73S/4qQaTwPXM2JBj6/6lI/byHH1jOpCzb0zDbfytgA9LcN/hzb2l7WQSQnxITOVx5un/wGw==",
|
||||
"funding": [
|
||||
"https://github.com/sponsors/broofa",
|
||||
"https://github.com/sponsors/ctavan"
|
||||
],
|
||||
"license": "MIT",
|
||||
"bin": {
|
||||
"uuid": "dist-node/bin/uuid"
|
||||
}
|
||||
},
|
||||
"node_modules/which": {
|
||||
"version": "2.0.2",
|
||||
"resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz",
|
||||
"integrity": "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA==",
|
||||
"license": "ISC",
|
||||
"dependencies": {
|
||||
"isexe": "^2.0.0"
|
||||
},
|
||||
"bin": {
|
||||
"node-which": "bin/node-which"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 8"
|
||||
}
|
||||
},
|
||||
"node_modules/yaml": {
|
||||
"version": "2.9.0",
|
||||
"resolved": "https://registry.npmjs.org/yaml/-/yaml-2.9.0.tgz",
|
||||
"integrity": "sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA==",
|
||||
"license": "ISC",
|
||||
"bin": {
|
||||
"yaml": "bin.mjs"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 14.6"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/eemeli"
|
||||
}
|
||||
},
|
||||
"node_modules/zod": {
|
||||
"version": "4.1.8",
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/colinhacks"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,164 +0,0 @@
|
||||
---
|
||||
name: audit-reasoning-options
|
||||
description: Audit or write models.dev reasoning_options in provider TOML files and reasoning-option PRs. Use when verifying toggle, effort, budget_tokens, provider reasoning controls, or citations.
|
||||
---
|
||||
|
||||
# Audit Reasoning Options
|
||||
|
||||
Use this workflow to add or review `reasoning_options` for a specific provider. Treat these fields as provider capabilities, not provider-agnostic model facts.
|
||||
|
||||
Provider capability means the inference service's accepted HTTP request surface. It does not mean the controls exposed by the repository's configured npm package, a preferred SDK, or a typed client wrapper.
|
||||
|
||||
## Available Options
|
||||
|
||||
The schema in `packages/core/src/schema.ts` supports:
|
||||
|
||||
```toml
|
||||
[[reasoning_options]]
|
||||
type = "toggle"
|
||||
|
||||
[[reasoning_options]]
|
||||
type = "effort"
|
||||
values = ["low", "medium", "high"]
|
||||
|
||||
[[reasoning_options]]
|
||||
type = "budget_tokens"
|
||||
min = 1_024
|
||||
max = 32_000
|
||||
```
|
||||
|
||||
- `toggle`: The provider offers an explicit way to switch reasoning on and off for the same model ID.
|
||||
- `effort`: The provider accepts one or more discrete effort values. Schema values are `null`, `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`, and `default`.
|
||||
- `budget_tokens`: The provider accepts a numeric reasoning-token budget. `min` and `max` are optional and must only be included when verified.
|
||||
- `reasoning_options = []`: The model reasons, but no user-selectable control was verified through this provider.
|
||||
- Omitted `reasoning_options`: No provider-specific claim has been authored. Do not treat omission as equivalent to an audited empty list.
|
||||
|
||||
An option describes a control exposed to a caller. Do not add an option merely because a model reasons internally or another provider exposes that control.
|
||||
|
||||
## Evidence Standard
|
||||
|
||||
Use evidence in this order:
|
||||
|
||||
1. The provider's current API reference or model documentation.
|
||||
2. The provider's raw OpenAPI schema, compatibility endpoint documentation, model endpoint metadata, or playground request payload.
|
||||
3. A reproducible request against the provider API, including a negative control with an invalid value where practical.
|
||||
4. The provider's official SDK source, but only as positive evidence for requests it emits.
|
||||
5. The upstream model developer's documentation.
|
||||
6. High-quality secondary sources only as supporting context.
|
||||
|
||||
Provider documentation proves what the provider accepts. Upstream documentation proves what the model can support, but cannot by itself prove that a gateway forwards or exposes the control.
|
||||
|
||||
An SDK can prove support when it emits a field. An SDK's omission, type restriction, or missing convenience option does not prove the inference API rejects that field. Before removing a control because an SDK cannot express it, inspect raw HTTP docs, compatibility base URLs, passthrough guarantees, migration guides, and direct API behavior.
|
||||
|
||||
Prefer versioned or model-specific documentation over generic examples. Record the access date when a page is mutable or unversioned.
|
||||
|
||||
## Audit Workflow
|
||||
|
||||
1. Read the provider configuration to identify the API base URL and protocol. Record the SDK only as one possible client.
|
||||
2. Inspect the PR diff and list every changed model with its exact proposed options.
|
||||
3. Group models by API family or request adapter, not only by model developer.
|
||||
4. Locate provider documentation for reasoning request fields and model-specific restrictions.
|
||||
5. Check every raw compatibility endpoint the inference provider advertises, such as OpenAI-, Anthropic-, or provider-compatible base URLs. Existing calls working unchanged is positive evidence that native reasoning fields are accepted.
|
||||
6. Cross-check upstream model documentation for supported values and ranges after establishing provider passthrough or translation.
|
||||
7. Test the provider API when credentials are already available and documentation is incomplete. Never print credentials.
|
||||
8. Compare each TOML claim independently: toggle, each effort value, budget support, minimum, and maximum.
|
||||
9. Remove any claim that lacks inference-provider evidence. Do not remove it merely because one SDK lacks a type or helper.
|
||||
10. Run `bun validate` and `git diff --check`.
|
||||
11. Update the PR body with citations, request-field details, audit conclusions, and validation commands.
|
||||
|
||||
## Toggle Verification
|
||||
|
||||
Only add `toggle` if all of these are true:
|
||||
|
||||
- The same provider model ID can run with reasoning enabled and disabled.
|
||||
- The caller controls the state through a documented or reproduced request.
|
||||
- The exact field and values are known.
|
||||
|
||||
Examples of possible controls include `thinking.type = "enabled" | "disabled"`, `enable_thinking = true | false`, a documented `reasoning` object, or a provider-defined prompt switch such as `/think` and `/no_think`.
|
||||
|
||||
The following do not prove a toggle:
|
||||
|
||||
- Separate thinking and non-thinking model IDs.
|
||||
- Omitting a reasoning budget when omission selects an automatic budget.
|
||||
- Setting effort to `low` unless the provider says it disables reasoning.
|
||||
- A model card saying the model is hybrid without provider request documentation.
|
||||
- A provider UI switch when its API payload cannot be identified.
|
||||
|
||||
For every proposed toggle, write this sentence before accepting it:
|
||||
|
||||
> `<provider model ID>` toggles reasoning with `<request path>` set to `<enabled value>` or `<disabled value>`.
|
||||
|
||||
If that sentence cannot be completed and cited or reproduced, do not claim `toggle`.
|
||||
|
||||
## Effort Verification
|
||||
|
||||
Verify every value separately. Do not copy the schema's full enum into a model.
|
||||
|
||||
- For an OpenAI-compatible API, `low`, `medium`, and `high` are a useful investigation baseline, not proof.
|
||||
- Require explicit evidence for `null`, `none`, `minimal`, `xhigh`, `max`, and `default`.
|
||||
- Check model-specific differences. A generic gateway enum may be rejected or ignored by some routed models.
|
||||
- Distinguish accepted values from meaningful values. If the gateway silently ignores a field, it is not a supported control.
|
||||
- Preserve JSON `null` as TOML `null`, not the string `"null"`, when evidence requires a null value.
|
||||
|
||||
When practical, send one valid request per claimed value and one invalid value. A structured `400` for the invalid value makes silent field dropping less likely.
|
||||
|
||||
## Budget Verification
|
||||
|
||||
`budget_tokens` is an abstract models.dev capability; providers may spell it `reasoning.max_tokens`, `thinking.budget_tokens`, `thinkingBudget`, or another field.
|
||||
|
||||
- Cite the provider's actual request path.
|
||||
- Verify that the field controls reasoning tokens rather than total output tokens.
|
||||
- Do not infer `max` from `limit.output`, context length, or an upstream provider's limit.
|
||||
- Do not infer a provider minimum from an SDK default.
|
||||
- Omit unverified bounds while retaining verified budget support.
|
||||
- Check whether zero or a negative sentinel disables reasoning. If so, verify whether this also proves `toggle` for that model.
|
||||
- Check constraints relating budget to `max_tokens` or total output.
|
||||
|
||||
## API Testing
|
||||
|
||||
Use existing credentials only when permitted and necessary. Keep secrets out of commands, logs, files, PR bodies, and chat output.
|
||||
|
||||
For each control, prefer this matrix:
|
||||
|
||||
| Request | Expected evidence |
|
||||
| --- | --- |
|
||||
| No reasoning field | Establishes default behavior |
|
||||
| Each claimed valid value | Successful response or documented acceptance |
|
||||
| Explicit disabled value | Proves toggle-off behavior |
|
||||
| One invalid value | Structured rejection rather than silent dropping |
|
||||
| Boundary and adjacent value | Supports a claimed minimum or maximum |
|
||||
|
||||
Acceptance alone is weak when an OpenAI-compatible gateway ignores unknown fields. Inspect returned metadata, reasoning content, usage fields, or error behavior where available.
|
||||
|
||||
## Citations
|
||||
|
||||
Put citations in the PR body, not TOML comments. TOML model files should remain data-only unless the repository establishes another convention.
|
||||
|
||||
Use direct links to the narrowest authoritative section. For each link, state exactly what it proves:
|
||||
|
||||
```markdown
|
||||
## Evidence
|
||||
|
||||
- [Provider reasoning API](https://example.com/api/reasoning) documents
|
||||
`reasoning_effort` values `low`, `medium`, and `high`.
|
||||
- [Provider model page](https://example.com/models/foo) documents that
|
||||
`thinking.type = "disabled"` turns reasoning off for `foo`.
|
||||
- [Upstream model documentation](https://example.com/upstream/foo) confirms
|
||||
the model-native budget range; provider requests at both boundaries succeeded.
|
||||
```
|
||||
|
||||
Do not cite a search-results page, an AI-generated summary, or a generic upstream page for a provider-specific claim. If evidence comes from authenticated endpoint metadata or testing, describe the endpoint, date, request field, result, and negative control without including credentials or sensitive response data.
|
||||
|
||||
## PR Audit Output
|
||||
|
||||
For each audited PR, report:
|
||||
|
||||
- Models and proposed options.
|
||||
- Verdict for every option: verified, corrected, or removed.
|
||||
- Exact toggle mechanism, when applicable.
|
||||
- Provider-level citations and what each proves.
|
||||
- Upstream citations used only for model-specific constraints.
|
||||
- Tests performed and their limitations.
|
||||
- Final validation result.
|
||||
|
||||
If documentation is ambiguous, state the ambiguity and use the least permissive metadata supported by evidence.
|
||||
@@ -1,18 +0,0 @@
|
||||
name = "Claude Fable 5"
|
||||
family = "claude-fable"
|
||||
release_date = "2026-06-09"
|
||||
last_updated = "2026-06-09"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
knowledge = "2026-01-31"
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -1,19 +0,0 @@
|
||||
name = "Command A Plus"
|
||||
family = "command-a"
|
||||
release_date = "2026-05-20"
|
||||
last_updated = "2026-06-09"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
knowledge = "2025-04-01"
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
structured_output = true
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
@@ -1,7 +1,7 @@
|
||||
name = "Command R7B"
|
||||
family = "command-r"
|
||||
release_date = "2024-12-02"
|
||||
last_updated = "2024-12-02"
|
||||
release_date = "2024-02-27"
|
||||
last_updated = "2024-02-27"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name = "Gemini 2.5 Flash"
|
||||
family = "gemini-flash"
|
||||
release_date = "2025-06-17"
|
||||
last_updated = "2025-06-17"
|
||||
release_date = "2025-03-20"
|
||||
last_updated = "2025-06-05"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name = "Gemini 2.5 Pro"
|
||||
family = "gemini-pro"
|
||||
release_date = "2025-06-17"
|
||||
last_updated = "2025-06-17"
|
||||
release_date = "2025-03-20"
|
||||
last_updated = "2025-06-05"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
|
||||
@@ -1,22 +0,0 @@
|
||||
name = "Gemma 4 E2B IT"
|
||||
family = "gemma"
|
||||
release_date = "2026-04-02"
|
||||
last_updated = "2026-04-02"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 8_192
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "audio"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/google/gemma-4-E2B-it"
|
||||
@@ -1,22 +0,0 @@
|
||||
name = "Gemma 4 E4B IT"
|
||||
family = "gemma"
|
||||
release_date = "2026-04-02"
|
||||
last_updated = "2026-04-02"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 8_192
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "audio"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/google/gemma-4-E4B-it"
|
||||
@@ -6,7 +6,7 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 512_000
|
||||
@@ -15,7 +15,3 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/MiniMaxAI/MiniMax-M3"
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name = "Mistral Large 2.1"
|
||||
family = "mistral-large"
|
||||
release_date = "2024-11-18"
|
||||
last_updated = "2024-11-18"
|
||||
release_date = "2024-11-01"
|
||||
last_updated = "2024-11-04"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
name = "Kimi K2.5"
|
||||
family = "kimi-k2"
|
||||
family = "kimi-k2.5"
|
||||
release_date = "2026-01"
|
||||
last_updated = "2026-01"
|
||||
attachment = false
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
name = "Kimi K2.6"
|
||||
family = "kimi-k2"
|
||||
family = "kimi-k2.6"
|
||||
release_date = "2026-04-21"
|
||||
last_updated = "2026-04-21"
|
||||
attachment = true
|
||||
|
||||
@@ -1,23 +0,0 @@
|
||||
name = "Kimi K2.7 Code Highspeed"
|
||||
family = "kimi-k2"
|
||||
release_date = "2026-06-12"
|
||||
last_updated = "2026-06-12"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 262_144
|
||||
output = 262_144
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
@@ -1,23 +0,0 @@
|
||||
name = "Kimi K2.7 Code"
|
||||
family = "kimi-k2"
|
||||
release_date = "2026-06-12"
|
||||
last_updated = "2026-06-12"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 262_144
|
||||
output = 262_144
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
@@ -1,17 +0,0 @@
|
||||
name = "GPT-Image-1.5"
|
||||
family = "gpt-image"
|
||||
release_date = "2025-11-25"
|
||||
last_updated = "2025-11-25"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 0
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text", "image"]
|
||||
@@ -1,17 +0,0 @@
|
||||
name = "GPT-Image-1"
|
||||
family = "gpt-image"
|
||||
release_date = "2025-04-24"
|
||||
last_updated = "2025-04-24"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 0
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["image"]
|
||||
@@ -1,17 +0,0 @@
|
||||
name = "GPT-Image-2"
|
||||
family = "gpt-image"
|
||||
release_date = "2026-04-21"
|
||||
last_updated = "2026-04-21"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 0
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["image"]
|
||||
@@ -1,22 +0,0 @@
|
||||
name = "GPT OSS 120B"
|
||||
family = "gpt-oss"
|
||||
release_date = "2025-08-05"
|
||||
last_updated = "2025-08-05"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/openai/gpt-oss-120b"
|
||||
@@ -1,22 +0,0 @@
|
||||
name = "GPT OSS Safeguard 120B"
|
||||
family = "gpt-oss"
|
||||
release_date = "2025-10-29"
|
||||
last_updated = "2025-10-29"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/openai/gpt-oss-safeguard-120b"
|
||||
@@ -1,16 +0,0 @@
|
||||
name = "Whisper Large v3 Turbo"
|
||||
family = "whisper"
|
||||
release_date = "2024-10-01"
|
||||
last_updated = "2024-10-01"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 448
|
||||
output = 448
|
||||
|
||||
[modalities]
|
||||
input = ["audio"]
|
||||
output = ["text"]
|
||||
@@ -1,16 +0,0 @@
|
||||
name = "Whisper 3 Large"
|
||||
family = "whisper"
|
||||
release_date = "2024-10-01"
|
||||
last_updated = "2024-10-01"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 448
|
||||
output = 4_096
|
||||
|
||||
[modalities]
|
||||
input = ["audio"]
|
||||
output = ["text"]
|
||||
@@ -6,11 +6,10 @@ attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
context = 2_000_000
|
||||
output = 30_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -6,11 +6,10 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
context = 2_000_000
|
||||
output = 30_000
|
||||
|
||||
[modalities]
|
||||
|
||||
@@ -6,7 +6,6 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
|
||||
@@ -1,22 +0,0 @@
|
||||
name = "MiMo-V2.5-Pro-UltraSpeed"
|
||||
family = "mimo"
|
||||
release_date = "2026-06-08"
|
||||
last_updated = "2026-06-09"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
knowledge = "2024-12"
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 131_072
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro-FP4-DFlash"
|
||||
@@ -1,7 +1,7 @@
|
||||
name = "GLM-5.1"
|
||||
family = "glm"
|
||||
release_date = "2026-04-07"
|
||||
last_updated = "2026-04-07"
|
||||
release_date = "2026-03-27"
|
||||
last_updated = "2026-03-27"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name = "GLM-5"
|
||||
family = "glm"
|
||||
release_date = "2026-02-12"
|
||||
last_updated = "2026-02-12"
|
||||
release_date = "2026-02-11"
|
||||
last_updated = "2026-02-11"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
|
||||
+2
-5
@@ -18,14 +18,11 @@
|
||||
"test": "bun test",
|
||||
"validate": "bun ./packages/core/script/validate.ts",
|
||||
"compare:migrations": "bun ./packages/core/script/compare-model-migrations.ts",
|
||||
"baseten:sync": "bun ./packages/core/script/sync-models.ts baseten",
|
||||
"cloudflare:sync": "bun ./packages/core/script/sync-models.ts cloudflare-workers-ai",
|
||||
"chutes:sync": "bun ./packages/core/script/sync-models.ts chutes",
|
||||
"chutes:generate": "bun ./packages/core/script/generate-chutes.ts",
|
||||
"databricks:generate": "bun ./packages/core/script/generate-databricks.ts",
|
||||
"helicone:generate": "bun ./packages/core/script/generate-helicone.ts",
|
||||
"huggingface:sync": "bun ./packages/core/script/sync-models.ts huggingface",
|
||||
"llmgateway:sync": "bun ./packages/core/script/sync-models.ts llmgateway",
|
||||
"venice:sync": "bun ./packages/core/script/sync-models.ts venice",
|
||||
"venice:generate": "bun ./packages/core/script/generate-venice.ts",
|
||||
"vercel:generate": "bun ./packages/core/script/sync-models.ts vercel",
|
||||
"wandb:generate": "bun ./packages/core/script/generate-wandb.ts",
|
||||
"digitalocean:generate": "bun ./packages/core/script/generate-digitalocean.ts",
|
||||
|
||||
@@ -0,0 +1,888 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
/**
|
||||
* Generates Chutes model TOML files from the Chutes LLM API.
|
||||
*
|
||||
* Flags:
|
||||
* --dry-run: Preview changes without writing files
|
||||
* --new-only: Only create new models, skip updating existing ones
|
||||
* --keep-orphans: Don't delete TOML files for models no longer in the API
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { existsSync, readFileSync } from "node:fs";
|
||||
import { mkdir } from "node:fs/promises";
|
||||
import { ModelFamilyValues } from "../src/family.js";
|
||||
|
||||
const API_ENDPOINT = "https://llm.chutes.ai/v1/models";
|
||||
const MODEL_METADATA_DIR = path.join(import.meta.dirname, "..", "..", "..", "models");
|
||||
|
||||
const CHUTES_ORG_TO_MODEL_PROVIDER: Record<string, string | undefined> = {
|
||||
"MiniMaxAI": "minimax",
|
||||
"Qwen": "alibaba",
|
||||
"XiaomiMiMo": "xiaomi",
|
||||
"deepseek-ai": "deepseek",
|
||||
"google": "google",
|
||||
"moonshotai": "moonshotai",
|
||||
"openai": "openai",
|
||||
"zai-org": "zhipuai",
|
||||
};
|
||||
|
||||
const BASE_MODEL_ALIASES: Record<string, string | undefined> = {
|
||||
"google/gemma-4-31B-turbo-TEE": "google/gemma-4-31b-it",
|
||||
"Qwen/Qwen3-235B-A22B-Instruct-2507-TEE": "alibaba/qwen3-235b-a22b",
|
||||
};
|
||||
|
||||
const modelMetadataByID = new Map<string, Record<string, unknown>>();
|
||||
|
||||
enum SkipZeroFields {
|
||||
LimitContext = "limit.context",
|
||||
LimitOutput = "limit.output",
|
||||
}
|
||||
|
||||
const Pricing = z.object({
|
||||
prompt: z.number().optional(),
|
||||
completion: z.number().optional(),
|
||||
input_cache_read: z.number().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const ChutesModel = z.object({
|
||||
id: z.string(),
|
||||
created: z.number(),
|
||||
pricing: Pricing.optional(),
|
||||
context_length: z.number().optional(),
|
||||
max_output_length: z.number().optional(),
|
||||
max_model_len: z.number().optional(),
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
supported_features: z.array(z.string()).optional(),
|
||||
supported_sampling_parameters: z.array(z.string()).optional(),
|
||||
quantization: z.string().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const ChutesResponse = z.object({
|
||||
data: z.array(ChutesModel),
|
||||
}).passthrough();
|
||||
|
||||
interface ExistingModel {
|
||||
base_model?: string;
|
||||
base_model_omit?: string[];
|
||||
name?: string;
|
||||
family?: string;
|
||||
attachment?: boolean;
|
||||
reasoning?: boolean;
|
||||
tool_call?: boolean;
|
||||
structured_output?: boolean;
|
||||
temperature?: boolean;
|
||||
knowledge?: string;
|
||||
release_date?: string;
|
||||
last_updated?: string;
|
||||
open_weights?: boolean;
|
||||
interleaved?: boolean | { field: string };
|
||||
status?: string;
|
||||
cost?: {
|
||||
input?: number;
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
};
|
||||
limit?: {
|
||||
context?: number;
|
||||
output?: number;
|
||||
};
|
||||
modalities?: {
|
||||
input?: string[];
|
||||
output?: string[];
|
||||
};
|
||||
}
|
||||
|
||||
interface MergedModel {
|
||||
base_model?: string;
|
||||
base_model_omit?: string[];
|
||||
name: string;
|
||||
family?: string;
|
||||
attachment: boolean;
|
||||
reasoning: boolean;
|
||||
tool_call: boolean;
|
||||
structured_output: boolean;
|
||||
temperature: boolean;
|
||||
knowledge?: string;
|
||||
release_date: string;
|
||||
last_updated: string;
|
||||
open_weights: boolean;
|
||||
interleaved?: boolean | { field: string };
|
||||
status?: string;
|
||||
cost?: {
|
||||
input: number;
|
||||
output: number;
|
||||
cache_read?: number;
|
||||
};
|
||||
limit: {
|
||||
context: number;
|
||||
output: number;
|
||||
};
|
||||
modalities: {
|
||||
input: string[];
|
||||
output: string[];
|
||||
};
|
||||
}
|
||||
|
||||
interface Changes {
|
||||
field: string;
|
||||
oldValue: string;
|
||||
newValue: string;
|
||||
}
|
||||
|
||||
// ── Utility functions ────────────────────────────────────────────────
|
||||
|
||||
function timestampToDate(timestamp: number): string {
|
||||
const date = new Date(timestamp * 1000);
|
||||
return date.toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function getTodayDate(): string {
|
||||
return new Date().toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function formatNumber(n: number): string {
|
||||
if (n >= 1000) {
|
||||
return n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_");
|
||||
}
|
||||
return n.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Humanize a model ID into a readable name.
|
||||
* Strips the org prefix and replaces hyphens with spaces.
|
||||
* e.g. "Qwen/Qwen3-32B-TEE" → "Qwen3 32B TEE"
|
||||
*/
|
||||
function humanizeModelName(modelId: string): string {
|
||||
const parts = modelId.split("/");
|
||||
const modelPart = parts.at(-1) ?? modelId;
|
||||
return modelPart.replace(/-/g, " ");
|
||||
}
|
||||
|
||||
function modelMetadataPath(modelId: string): string {
|
||||
return path.join(MODEL_METADATA_DIR, `${modelId}.toml`);
|
||||
}
|
||||
|
||||
function modelMetadataExists(modelId: string): boolean {
|
||||
return existsSync(modelMetadataPath(modelId));
|
||||
}
|
||||
|
||||
function modelMetadata(modelId: string): Record<string, unknown> {
|
||||
let metadata = modelMetadataByID.get(modelId);
|
||||
if (metadata === undefined) {
|
||||
metadata = Bun.TOML.parse(
|
||||
readFileSync(modelMetadataPath(modelId), "utf8"),
|
||||
) as Record<string, unknown>;
|
||||
modelMetadataByID.set(modelId, metadata);
|
||||
}
|
||||
return metadata;
|
||||
}
|
||||
|
||||
function baseModelCandidates(modelId: string): string[] {
|
||||
const alias = BASE_MODEL_ALIASES[modelId];
|
||||
const [org, ...modelParts] = modelId.split("/");
|
||||
if (org === undefined || modelParts.length === 0) {
|
||||
return alias === undefined ? [] : [alias];
|
||||
}
|
||||
|
||||
const provider = CHUTES_ORG_TO_MODEL_PROVIDER[org];
|
||||
if (provider === undefined) {
|
||||
return alias === undefined ? [] : [alias];
|
||||
}
|
||||
|
||||
const rawModel = modelParts.join("/");
|
||||
if (!rawModel.endsWith("-TEE")) {
|
||||
return alias === undefined ? [] : [alias];
|
||||
}
|
||||
|
||||
const withoutTee = rawModel.slice(0, -"-TEE".length);
|
||||
const lower = withoutTee.toLowerCase();
|
||||
const normalized = [
|
||||
withoutTee,
|
||||
lower,
|
||||
lower.replace(/-(?:instruct|thinking)-\d{4}$/, ""),
|
||||
lower.replace(/-\d{4}$/, ""),
|
||||
lower.replace(/-turbo$/, "-it"),
|
||||
lower.replace(/-turbo$/, ""),
|
||||
];
|
||||
|
||||
return [...new Set([alias, ...normalized.map((candidate) => `${provider}/${candidate}`)])
|
||||
.values()].filter((candidate): candidate is string => candidate !== undefined);
|
||||
}
|
||||
|
||||
function resolveBaseModel(modelId: string, existing: ExistingModel | null): string | undefined {
|
||||
const candidates = [
|
||||
existing?.base_model,
|
||||
...baseModelCandidates(modelId),
|
||||
].filter((candidate): candidate is string => candidate !== undefined);
|
||||
|
||||
return candidates.find(modelMetadataExists);
|
||||
}
|
||||
|
||||
function resolveBaseModelOmit(
|
||||
baseModel: string | undefined,
|
||||
existing: ExistingModel | null,
|
||||
): string[] | undefined {
|
||||
const omit = new Set(existing?.base_model_omit ?? []);
|
||||
if (baseModel !== undefined) {
|
||||
const baseLimit = modelMetadata(baseModel).limit;
|
||||
if (
|
||||
isPlainObject(baseLimit) &&
|
||||
baseLimit.input !== undefined
|
||||
) {
|
||||
omit.add("limit.input");
|
||||
}
|
||||
}
|
||||
|
||||
return omit.size > 0 ? [...omit].sort() : undefined;
|
||||
}
|
||||
|
||||
// ── Family inference ───────────
|
||||
|
||||
function isSubstring(target: string, family: string): boolean {
|
||||
return target.toLowerCase().includes(family.toLowerCase());
|
||||
}
|
||||
|
||||
function matchesFamily(target: string, family: string): boolean {
|
||||
const targetLower = target.toLowerCase();
|
||||
const familyLower = family.toLowerCase();
|
||||
let familyIdx = 0;
|
||||
|
||||
for (let i = 0; i < targetLower.length && familyIdx < familyLower.length; i++) {
|
||||
if (targetLower[i] === familyLower[familyIdx]) {
|
||||
familyIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
return familyIdx === familyLower.length;
|
||||
}
|
||||
|
||||
function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
const sortedFamilies = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
|
||||
|
||||
// First pass: try exact substring matches
|
||||
for (const family of sortedFamilies) {
|
||||
if (isSubstring(modelId, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
if (isSubstring(modelName, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
// Second pass: fall back to subsequence matching
|
||||
for (const family of sortedFamilies) {
|
||||
if (matchesFamily(modelId, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
if (matchesFamily(modelName, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// ── Load existing TOML ───────────────────────────────────────────────
|
||||
|
||||
async function loadExistingModel(filePath: string): Promise<ExistingModel | null> {
|
||||
try {
|
||||
const file = Bun.file(filePath);
|
||||
if (!(await file.exists())) {
|
||||
return null;
|
||||
}
|
||||
const toml = await import(filePath, { with: { type: "toml" } }).then(
|
||||
(mod) => mod.default,
|
||||
);
|
||||
return toml as ExistingModel;
|
||||
} catch (e) {
|
||||
console.warn(`Warning: Failed to parse existing file ${filePath}:`, e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// ── Merge API data with existing TOML ────────────────────────────────
|
||||
|
||||
function mergeModel(
|
||||
apiModel: z.infer<typeof ChutesModel>,
|
||||
existing: ExistingModel | null,
|
||||
): MergedModel {
|
||||
const features = new Set(apiModel.supported_features ?? []);
|
||||
const samplingParams = new Set(apiModel.supported_sampling_parameters ?? []);
|
||||
const inputMods = apiModel.input_modalities ?? ["text"];
|
||||
const outputMods = apiModel.output_modalities ?? ["text"];
|
||||
|
||||
// Capabilities from API features
|
||||
const hasAttachment = inputMods.some((m) =>
|
||||
m === "image" || m === "video" || m === "pdf",
|
||||
);
|
||||
const hasReasoning = features.has("reasoning");
|
||||
const hasToolCall = features.has("tools");
|
||||
const hasStructuredOutput = features.has("structured_outputs");
|
||||
const hasTemperature = samplingParams.size > 0
|
||||
? samplingParams.has("temperature")
|
||||
: true; // default true if no sampling params info
|
||||
|
||||
// Preserve existing values when available (manually specified)
|
||||
const modelName = existing?.name ?? humanizeModelName(apiModel.id);
|
||||
const family = existing?.family ?? inferFamily(apiModel.id, modelName);
|
||||
const knowledge = existing?.knowledge;
|
||||
const interleaved = existing?.interleaved;
|
||||
const status = existing?.status;
|
||||
const baseModel = resolveBaseModel(apiModel.id, existing);
|
||||
const baseModelOmit = resolveBaseModelOmit(baseModel, existing);
|
||||
|
||||
// Release date: existing > API created timestamp > today
|
||||
const releaseDate = existing?.release_date
|
||||
?? timestampToDate(apiModel.created)
|
||||
?? getTodayDate();
|
||||
|
||||
// Context limit: prefer context_length, fallback to max_model_len
|
||||
const apiContext = apiModel.context_length ?? apiModel.max_model_len ?? 0;
|
||||
const contextLimit = apiContext > 0
|
||||
? apiContext
|
||||
: (existing?.limit?.context ?? 0);
|
||||
|
||||
// Output limit: prefer max_output_length, fallback to existing
|
||||
const apiOutput = apiModel.max_output_length ?? 0;
|
||||
const outputLimit = apiOutput > 0
|
||||
? apiOutput
|
||||
: (existing?.limit?.output ?? 0);
|
||||
|
||||
const merged: MergedModel = {
|
||||
...(baseModel !== undefined && { base_model: baseModel }),
|
||||
...(baseModelOmit !== undefined && { base_model_omit: baseModelOmit }),
|
||||
name: modelName,
|
||||
family,
|
||||
attachment: hasAttachment,
|
||||
reasoning: hasReasoning,
|
||||
tool_call: hasToolCall,
|
||||
temperature: hasTemperature,
|
||||
structured_output: hasStructuredOutput,
|
||||
release_date: releaseDate,
|
||||
last_updated: getTodayDate(),
|
||||
open_weights: true, // Chutes hosts open-weight models
|
||||
...(knowledge && { knowledge }),
|
||||
...(interleaved !== undefined && { interleaved }),
|
||||
...(status && { status }),
|
||||
limit: {
|
||||
context: contextLimit,
|
||||
output: outputLimit,
|
||||
},
|
||||
modalities: {
|
||||
input: inputMods,
|
||||
output: outputMods,
|
||||
},
|
||||
};
|
||||
|
||||
// Cost: API values are already in USD per 1M tokens — use directly
|
||||
if (apiModel.pricing) {
|
||||
const inputPrice = apiModel.pricing.prompt;
|
||||
const outputPrice = apiModel.pricing.completion;
|
||||
const cacheReadPrice = apiModel.pricing.input_cache_read;
|
||||
|
||||
if (inputPrice !== undefined && outputPrice !== undefined) {
|
||||
merged.cost = {
|
||||
input: inputPrice,
|
||||
output: outputPrice,
|
||||
...(cacheReadPrice !== undefined && { cache_read: cacheReadPrice }),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return merged;
|
||||
}
|
||||
|
||||
// ── TOML formatting ──────────────────────────────────────────────────
|
||||
|
||||
function formatToml(model: MergedModel): string {
|
||||
if (model.base_model !== undefined) {
|
||||
return formatBaseModelToml(model);
|
||||
}
|
||||
|
||||
return formatFullToml(model);
|
||||
}
|
||||
|
||||
function formatFullToml(model: MergedModel): string {
|
||||
const lines: string[] = [];
|
||||
|
||||
lines.push(`# Auto-generated by generate-chutes.ts — do not edit pricing, limits, or capabilities.`);
|
||||
lines.push(`# Manual overrides preserved on re-run: name, family, knowledge, interleaved, status`);
|
||||
lines.push(`name = "${model.name.replace(/"/g, '\\"')}"`);
|
||||
if (model.family) {
|
||||
lines.push(`family = "${model.family}"`);
|
||||
}
|
||||
lines.push(`release_date = "${model.release_date}"`);
|
||||
lines.push(`last_updated = "${model.last_updated}"`);
|
||||
lines.push(`attachment = ${model.attachment}`);
|
||||
lines.push(`reasoning = ${model.reasoning}`);
|
||||
lines.push(`temperature = ${model.temperature}`);
|
||||
lines.push(`tool_call = ${model.tool_call}`);
|
||||
if (model.structured_output) {
|
||||
lines.push(`structured_output = ${model.structured_output}`);
|
||||
}
|
||||
lines.push(`open_weights = ${model.open_weights}`);
|
||||
if (model.knowledge) {
|
||||
lines.push(`knowledge = "${model.knowledge}"`);
|
||||
}
|
||||
if (model.status) {
|
||||
lines.push(`status = "${model.status}"`);
|
||||
}
|
||||
|
||||
if (model.cost) {
|
||||
lines.push("");
|
||||
lines.push(`[cost]`);
|
||||
lines.push(`input = ${model.cost.input}`);
|
||||
lines.push(`output = ${model.cost.output}`);
|
||||
if (model.cost.cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${model.cost.cache_read}`);
|
||||
}
|
||||
}
|
||||
|
||||
lines.push("");
|
||||
lines.push(`[limit]`);
|
||||
lines.push(`context = ${formatNumber(model.limit.context)}`);
|
||||
lines.push(`output = ${formatNumber(model.limit.output)}`);
|
||||
|
||||
lines.push("");
|
||||
lines.push(`[modalities]`);
|
||||
lines.push(`input = [${model.modalities.input.map((m) => `"${m}"`).join(", ")}]`);
|
||||
lines.push(`output = [${model.modalities.output.map((m) => `"${m}"`).join(", ")}]`);
|
||||
|
||||
if (model.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (model.interleaved === true) {
|
||||
lines.push(`interleaved = true`);
|
||||
} else if (typeof model.interleaved === "object") {
|
||||
lines.push(`[interleaved]`);
|
||||
lines.push(`field = "${model.interleaved.field}"`);
|
||||
}
|
||||
}
|
||||
|
||||
return lines.join("\n") + "\n";
|
||||
}
|
||||
|
||||
function formatBaseModelToml(model: MergedModel): string {
|
||||
const lines: string[] = [];
|
||||
const overrides = baseModelOverrides(model);
|
||||
|
||||
lines.push(`# Auto-generated by generate-chutes.ts — do not edit pricing, limits, or capabilities.`);
|
||||
lines.push(`# Manual overrides preserved on re-run: name, family, knowledge, interleaved, status`);
|
||||
lines.push(`base_model = "${model.base_model}"`);
|
||||
if (model.base_model_omit !== undefined) {
|
||||
lines.push(
|
||||
`base_model_omit = [${model.base_model_omit.map((item) => `"${item}"`).join(", ")}]`,
|
||||
);
|
||||
}
|
||||
if (overrides.name !== undefined) {
|
||||
lines.push(`name = "${String(overrides.name).replace(/"/g, '\\"')}"`);
|
||||
}
|
||||
for (const field of [
|
||||
"attachment",
|
||||
"reasoning",
|
||||
"structured_output",
|
||||
"temperature",
|
||||
"tool_call",
|
||||
"open_weights",
|
||||
] as const) {
|
||||
const value = overrides[field];
|
||||
if (value !== undefined) {
|
||||
lines.push(`${field} = ${value}`);
|
||||
}
|
||||
}
|
||||
if (overrides.knowledge !== undefined) {
|
||||
lines.push(`knowledge = "${overrides.knowledge}"`);
|
||||
}
|
||||
if (overrides.status !== undefined) {
|
||||
lines.push(`status = "${overrides.status}"`);
|
||||
}
|
||||
|
||||
if (overrides.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (overrides.interleaved === true) {
|
||||
lines.push(`interleaved = true`);
|
||||
} else if (isPlainObject(overrides.interleaved)) {
|
||||
lines.push(`[interleaved]`);
|
||||
lines.push(`field = "${overrides.interleaved.field}"`);
|
||||
}
|
||||
}
|
||||
|
||||
if (model.cost) {
|
||||
lines.push("");
|
||||
lines.push(`[cost]`);
|
||||
lines.push(`input = ${model.cost.input}`);
|
||||
lines.push(`output = ${model.cost.output}`);
|
||||
if (model.cost.cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${model.cost.cache_read}`);
|
||||
}
|
||||
}
|
||||
|
||||
lines.push("");
|
||||
lines.push(`[limit]`);
|
||||
lines.push(`context = ${formatNumber(model.limit.context)}`);
|
||||
lines.push(`output = ${formatNumber(model.limit.output)}`);
|
||||
|
||||
if (overrides.modalities !== undefined && isPlainObject(overrides.modalities)) {
|
||||
const input = overrides.modalities.input;
|
||||
const output = overrides.modalities.output;
|
||||
if (Array.isArray(input) && Array.isArray(output)) {
|
||||
lines.push("");
|
||||
lines.push(`[modalities]`);
|
||||
lines.push(`input = [${input.map((m) => `"${m}"`).join(", ")}]`);
|
||||
lines.push(`output = [${output.map((m) => `"${m}"`).join(", ")}]`);
|
||||
}
|
||||
}
|
||||
|
||||
return lines.join("\n") + "\n";
|
||||
}
|
||||
|
||||
function baseModelOverrides(model: MergedModel): Record<string, unknown> {
|
||||
if (model.base_model === undefined) {
|
||||
return {};
|
||||
}
|
||||
|
||||
const metadata = modelMetadata(model.base_model);
|
||||
const values: Record<string, unknown> = {
|
||||
name: model.name,
|
||||
attachment: model.attachment,
|
||||
reasoning: model.reasoning,
|
||||
structured_output:
|
||||
model.structured_output || metadata.structured_output === true
|
||||
? model.structured_output
|
||||
: undefined,
|
||||
temperature: model.temperature,
|
||||
tool_call: model.tool_call,
|
||||
knowledge: model.knowledge,
|
||||
open_weights: model.open_weights,
|
||||
status: model.status,
|
||||
interleaved: model.interleaved,
|
||||
modalities: model.modalities,
|
||||
};
|
||||
|
||||
return Object.fromEntries(
|
||||
Object.entries(values)
|
||||
.map(([key, value]) => [key, inheritedOverride(value, metadata[key])])
|
||||
.filter(([, value]) => value !== undefined),
|
||||
);
|
||||
}
|
||||
|
||||
function inheritedOverride(value: unknown, inherited: unknown): unknown {
|
||||
if (value === undefined) return undefined;
|
||||
if (sameInheritedValue(value, inherited)) return undefined;
|
||||
return stripUndefined(value);
|
||||
}
|
||||
|
||||
function stripUndefined(value: unknown): unknown {
|
||||
if (Array.isArray(value)) return value.map(stripUndefined);
|
||||
if (isPlainObject(value)) {
|
||||
return Object.fromEntries(
|
||||
Object.entries(value)
|
||||
.filter(([, item]) => item !== undefined)
|
||||
.map(([key, item]) => [key, stripUndefined(item)]),
|
||||
);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function sameInheritedValue(value: unknown, inherited: unknown): boolean {
|
||||
return stableInheritedValue(value) === stableInheritedValue(inherited);
|
||||
}
|
||||
|
||||
function stableInheritedValue(value: unknown): string {
|
||||
if (Array.isArray(value)) {
|
||||
const items = value.map(stableInheritedValue);
|
||||
const ordered = value.every((item) => item === null || typeof item !== "object")
|
||||
? items.sort()
|
||||
: items;
|
||||
return `[${ordered.join(",")}]`;
|
||||
}
|
||||
if (isPlainObject(value)) {
|
||||
return `{${Object.entries(value)
|
||||
.filter(([, item]) => item !== undefined)
|
||||
.sort(([a], [b]) => a.localeCompare(b))
|
||||
.map(([key, item]) => `${JSON.stringify(key)}:${stableInheritedValue(item)}`)
|
||||
.join(",")}}`;
|
||||
}
|
||||
return JSON.stringify(value);
|
||||
}
|
||||
|
||||
function isPlainObject(value: unknown): value is Record<string, unknown> {
|
||||
return value !== null && typeof value === "object" && !Array.isArray(value);
|
||||
}
|
||||
|
||||
// ── Change detection ─────────────────────────────────────────────────
|
||||
|
||||
function detectChanges(
|
||||
existing: ExistingModel | null,
|
||||
merged: MergedModel,
|
||||
): Changes[] {
|
||||
if (!existing) return [];
|
||||
|
||||
const changes: Changes[] = [];
|
||||
const EPSILON = 0.001;
|
||||
|
||||
const shouldSkipZero = (field: string, oldVal: unknown, newVal: unknown): boolean => {
|
||||
if (!Object.values(SkipZeroFields).includes(field as SkipZeroFields)) {
|
||||
return false;
|
||||
}
|
||||
return (typeof oldVal === "number" && oldVal === 0) || (typeof newVal === "number" && newVal === 0);
|
||||
};
|
||||
|
||||
const formatValue = (val: unknown): string => {
|
||||
if (typeof val === "number") return formatNumber(val);
|
||||
if (Array.isArray(val)) return `[${val.join(", ")}]`;
|
||||
if (val === undefined) return "(none)";
|
||||
return String(val);
|
||||
};
|
||||
|
||||
const isMaterialPriceDiff = (oldPrice: unknown, newPrice: unknown): boolean => {
|
||||
if (oldPrice === 0 && newPrice === undefined) return false;
|
||||
if (oldPrice !== undefined && newPrice !== undefined) {
|
||||
return Math.abs((oldPrice as number) - (newPrice as number)) > EPSILON;
|
||||
}
|
||||
return oldPrice !== newPrice;
|
||||
};
|
||||
|
||||
const compare = (field: string, oldVal: unknown, newVal: unknown) => {
|
||||
if (shouldSkipZero(field, oldVal, newVal)) return;
|
||||
|
||||
const isDiff = field.startsWith("cost.")
|
||||
? isMaterialPriceDiff(oldVal, newVal)
|
||||
: JSON.stringify(oldVal) !== JSON.stringify(newVal);
|
||||
|
||||
if (isDiff) {
|
||||
changes.push({
|
||||
field,
|
||||
oldValue: formatValue(oldVal),
|
||||
newValue: formatValue(newVal),
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
if (merged.base_model !== undefined) {
|
||||
const overrides = baseModelOverrides(merged);
|
||||
|
||||
compare("base_model", existing.base_model, merged.base_model);
|
||||
compare("base_model_omit", existing.base_model_omit, merged.base_model_omit);
|
||||
compare("name", existing.name, overrides.name);
|
||||
compare("attachment", existing.attachment, overrides.attachment);
|
||||
compare("reasoning", existing.reasoning, overrides.reasoning);
|
||||
compare("tool_call", existing.tool_call, overrides.tool_call);
|
||||
compare(
|
||||
"structured_output",
|
||||
existing.structured_output ?? false,
|
||||
overrides.structured_output ?? false,
|
||||
);
|
||||
compare("temperature", existing.temperature, overrides.temperature);
|
||||
compare("open_weights", existing.open_weights, overrides.open_weights);
|
||||
compare("knowledge", existing.knowledge, overrides.knowledge);
|
||||
compare("status", existing.status, overrides.status);
|
||||
compare("interleaved", existing.interleaved, overrides.interleaved);
|
||||
compare("cost.input", existing.cost?.input, merged.cost?.input);
|
||||
compare("cost.output", existing.cost?.output, merged.cost?.output);
|
||||
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
|
||||
compare("limit.context", existing.limit?.context, merged.limit.context);
|
||||
compare("limit.output", existing.limit?.output, merged.limit.output);
|
||||
if (isPlainObject(overrides.modalities)) {
|
||||
compare("modalities.input", existing.modalities?.input, overrides.modalities.input);
|
||||
compare("modalities.output", existing.modalities?.output, overrides.modalities.output);
|
||||
} else {
|
||||
compare("modalities.input", existing.modalities?.input, undefined);
|
||||
compare("modalities.output", existing.modalities?.output, undefined);
|
||||
}
|
||||
|
||||
return changes;
|
||||
}
|
||||
|
||||
compare("name", existing.name, merged.name);
|
||||
compare("base_model", existing.base_model, merged.base_model);
|
||||
compare("base_model_omit", existing.base_model_omit, merged.base_model_omit);
|
||||
compare("family", existing.family, merged.family);
|
||||
compare("attachment", existing.attachment, merged.attachment);
|
||||
compare("reasoning", existing.reasoning, merged.reasoning);
|
||||
compare("tool_call", existing.tool_call, merged.tool_call);
|
||||
compare("structured_output", existing.structured_output ?? false, merged.structured_output);
|
||||
compare("open_weights", existing.open_weights, merged.open_weights);
|
||||
compare("release_date", existing.release_date, merged.release_date);
|
||||
compare("cost.input", existing.cost?.input, merged.cost?.input);
|
||||
compare("cost.output", existing.cost?.output, merged.cost?.output);
|
||||
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
|
||||
compare("limit.context", existing.limit?.context, merged.limit.context);
|
||||
compare("limit.output", existing.limit?.output, merged.limit.output);
|
||||
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
|
||||
compare("modalities.output", existing.modalities?.output, merged.modalities.output);
|
||||
|
||||
return changes;
|
||||
}
|
||||
|
||||
// ── Main ─────────────────────────────────────────────────────────────
|
||||
|
||||
async function main() {
|
||||
const args = process.argv.slice(2);
|
||||
const dryRun = args.includes("--dry-run");
|
||||
const newOnly = args.includes("--new-only");
|
||||
const keepOrphans = args.includes("--keep-orphans");
|
||||
|
||||
const modelsDir = path.join(
|
||||
import.meta.dirname,
|
||||
"..",
|
||||
"..",
|
||||
"..",
|
||||
"providers",
|
||||
"chutes",
|
||||
"models",
|
||||
);
|
||||
|
||||
console.log(`${dryRun ? "[DRY RUN] " : ""}${newOnly ? "[NEW ONLY] " : ""}${keepOrphans ? "[KEEP ORPHANS] " : ""}Fetching Chutes models from API...`);
|
||||
|
||||
const res = await fetch(API_ENDPOINT);
|
||||
if (!res.ok) {
|
||||
console.error(`Failed to fetch API: ${res.status} ${res.statusText}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const json = await res.json();
|
||||
const parsed = ChutesResponse.safeParse(json);
|
||||
if (!parsed.success) {
|
||||
console.error("Invalid API response:", parsed.error.errors);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const apiModels = parsed.data.data;
|
||||
|
||||
// Scan existing TOML files
|
||||
const existingFiles = new Set<string>();
|
||||
try {
|
||||
for await (const file of new Bun.Glob("**/*.toml").scan({
|
||||
cwd: modelsDir,
|
||||
absolute: false,
|
||||
})) {
|
||||
existingFiles.add(file);
|
||||
}
|
||||
} catch {
|
||||
}
|
||||
|
||||
console.log(`Found ${apiModels.length} models in API, ${existingFiles.size} existing files\n`);
|
||||
|
||||
const apiModelIds = new Set<string>();
|
||||
|
||||
let created = 0;
|
||||
let updated = 0;
|
||||
let unchanged = 0;
|
||||
|
||||
for (const apiModel of apiModels) {
|
||||
const relativePath = `${apiModel.id}.toml`;
|
||||
const filePath = path.join(modelsDir, relativePath);
|
||||
const dirPath = path.dirname(filePath);
|
||||
|
||||
apiModelIds.add(relativePath);
|
||||
|
||||
const existing = await loadExistingModel(filePath);
|
||||
const merged = mergeModel(apiModel, existing);
|
||||
const tomlContent = formatToml(merged);
|
||||
|
||||
if (existing === null) {
|
||||
created++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would create: ${relativePath}`);
|
||||
console.log(` name = "${merged.name}"`);
|
||||
if (merged.family) {
|
||||
console.log(` family = "${merged.family}" (inferred)`);
|
||||
}
|
||||
console.log("");
|
||||
} else {
|
||||
await mkdir(dirPath, { recursive: true });
|
||||
await Bun.write(filePath, tomlContent);
|
||||
console.log(`Created: ${relativePath}`);
|
||||
}
|
||||
} else {
|
||||
if (newOnly) {
|
||||
unchanged++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const changes = detectChanges(existing, merged);
|
||||
const existingContent = await Bun.file(filePath).text();
|
||||
const formatChanged = existingContent !== tomlContent;
|
||||
|
||||
if (changes.length > 0 || formatChanged) {
|
||||
updated++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would update: ${relativePath}`);
|
||||
} else {
|
||||
await mkdir(dirPath, { recursive: true });
|
||||
await Bun.write(filePath, tomlContent);
|
||||
console.log(`Updated: ${relativePath}`);
|
||||
}
|
||||
for (const change of changes) {
|
||||
console.log(` ${change.field}: ${change.oldValue} → ${change.newValue}`);
|
||||
}
|
||||
if (changes.length === 0 && formatChanged) {
|
||||
console.log(` (format-only change)`);
|
||||
}
|
||||
console.log("");
|
||||
} else {
|
||||
unchanged++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Handle orphaned files (on disk but not in API)
|
||||
const orphaned: string[] = [];
|
||||
for (const file of existingFiles) {
|
||||
if (!apiModelIds.has(file)) {
|
||||
orphaned.push(file);
|
||||
const orphanPath = path.join(modelsDir, file);
|
||||
if (keepOrphans) {
|
||||
console.log(`Orphaned (kept): ${file}`);
|
||||
} else if (dryRun) {
|
||||
console.log(`[DRY RUN] Would delete: ${file}`);
|
||||
} else {
|
||||
await Bun.file(orphanPath).delete();
|
||||
console.log(`Deleted: ${file}`);
|
||||
|
||||
// Clean up empty parent directories
|
||||
const parentDir = path.dirname(orphanPath);
|
||||
try {
|
||||
const remaining = [];
|
||||
for await (const entry of new Bun.Glob("*").scan({ cwd: parentDir })) {
|
||||
remaining.push(entry);
|
||||
}
|
||||
if (remaining.length === 0) {
|
||||
const { rmdir } = await import("node:fs/promises");
|
||||
await rmdir(parentDir);
|
||||
console.log(` Removed empty directory: ${path.basename(parentDir)}/`);
|
||||
}
|
||||
} catch {
|
||||
// Directory not empty or other error, ignore
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log("");
|
||||
if (dryRun) {
|
||||
console.log(
|
||||
`Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} would be deleted`,
|
||||
);
|
||||
} else if (keepOrphans) {
|
||||
console.log(
|
||||
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned (kept)`,
|
||||
);
|
||||
} else {
|
||||
console.log(
|
||||
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} deleted`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
await main();
|
||||
@@ -26,7 +26,7 @@
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { mkdir } from "node:fs/promises";
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../src/family.js";
|
||||
import { ModelFamilyValues } from "../src/family.js";
|
||||
|
||||
const MODELS_API = "https://api.digitalocean.com/v2/gen-ai/models";
|
||||
const PRICING_API = "https://www.digitalocean.com/api/static-content/v1/products";
|
||||
@@ -142,7 +142,7 @@ const PRICING_NAME_MAP: Record<string, string> = {
|
||||
// DO-hosted
|
||||
"qwen3-32b": "alibaba-qwen3-32b",
|
||||
"minimax m2.5 (public preview)": "minimax-m2.5",
|
||||
"kimi k2.5": "kimi-k2",
|
||||
"kimi k2.5": "kimi-k2.5",
|
||||
"nvidia nemotron 3 super 120b (public preview)": "nvidia-nemotron-3-super-120b",
|
||||
"glm 5": "glm-5",
|
||||
};
|
||||
@@ -311,9 +311,6 @@ function formatNumber(n: number): string {
|
||||
}
|
||||
|
||||
function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
const kimiFamily = inferKimiFamily(modelId, modelName);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const sorted = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
|
||||
const targets = [modelId.toLowerCase(), modelName.toLowerCase()];
|
||||
for (const family of sorted) {
|
||||
|
||||
@@ -4,8 +4,6 @@ import { mkdir } from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { z } from "zod";
|
||||
|
||||
import { inferKimiFamily } from "../src/family.js";
|
||||
|
||||
// Friendli API endpoint
|
||||
const API_ENDPOINT = "https://api.friendli.ai/serverless/v1/models";
|
||||
|
||||
@@ -55,9 +53,6 @@ const familyPatterns: [RegExp, string][] = [
|
||||
];
|
||||
|
||||
function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
const kimiFamily = inferKimiFamily(modelId, modelName);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
for (const [pattern, family] of familyPatterns) {
|
||||
if (pattern.test(modelId) || pattern.test(modelName)) {
|
||||
return family;
|
||||
|
||||
@@ -0,0 +1,733 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { readdir } from "node:fs/promises";
|
||||
import { ModelFamilyValues } from "../src/family.js";
|
||||
|
||||
// Venice API endpoint
|
||||
const API_ENDPOINT = "https://api.venice.ai/api/v1/models?type=text";
|
||||
|
||||
// Zod schemas for API response validation
|
||||
const ReasoningEffort = z.enum(["none", "minimal", "low", "medium", "high", "xhigh", "max"]);
|
||||
type ReasoningOption = {
|
||||
type: "effort";
|
||||
values: Array<z.infer<typeof ReasoningEffort>>;
|
||||
};
|
||||
|
||||
const effort = (...values: ReasoningOption["values"]): ReasoningOption[] => [{ type: "effort", values }];
|
||||
|
||||
// Venice documents these model-specific values even where /models is stale or incomplete.
|
||||
// Source: https://docs.venice.ai/guides/features/reasoning-models
|
||||
export const REASONING_OVERRIDES: Record<string, ReasoningOption[]> = {
|
||||
"claude-opus-4-5": effort("low", "medium", "high"),
|
||||
"claude-opus-4-6": effort("low", "medium", "high", "max"),
|
||||
"claude-opus-4-6-fast": effort("low", "medium", "high", "max"),
|
||||
"claude-sonnet-4-5": effort("low", "medium", "high"),
|
||||
"claude-sonnet-4-6": effort("low", "medium", "high"),
|
||||
"gemini-3-flash-preview": effort("minimal", "low", "medium", "high"),
|
||||
"kimi-k2-5": effort("low", "medium", "high"),
|
||||
"openai-gpt-52": effort("none", "low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-52-codex": effort("low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-53-codex": effort("low", "medium", "high", "xhigh"),
|
||||
"qwen3-5-35b-a3b": effort("low", "medium", "high"),
|
||||
"zai-org-glm-5-1": [],
|
||||
|
||||
// Provisional until funded Venice probes can confirm that its proxy preserves
|
||||
// OpenAI's current controls. Sources:
|
||||
// https://developers.openai.com/api/docs/models/gpt-5.4
|
||||
// https://developers.openai.com/api/docs/models/gpt-5.4-pro
|
||||
// https://developers.openai.com/api/docs/guides/reasoning
|
||||
"openai-gpt-54": effort("none", "low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-54-mini": effort("none", "low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-54-pro": effort("medium", "high", "xhigh"),
|
||||
"openai-gpt-55": effort("none", "low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-55-pro": effort("medium", "high", "xhigh"),
|
||||
};
|
||||
|
||||
const Capabilities = z
|
||||
.object({
|
||||
optimizedForCode: z.boolean().optional(),
|
||||
quantization: z.string().optional(),
|
||||
supportsAudioInput: z.boolean().optional(),
|
||||
supportsFunctionCalling: z.boolean().optional(),
|
||||
supportsLogProbs: z.boolean().optional(),
|
||||
supportsReasoning: z.boolean().optional(),
|
||||
supportsReasoningEffort: z.boolean().optional(),
|
||||
reasoningEffortOptions: z.array(ReasoningEffort).optional(),
|
||||
defaultReasoningEffort: ReasoningEffort.optional(),
|
||||
supportsResponseSchema: z.boolean().optional(),
|
||||
supportsVideoInput: z.boolean().optional(),
|
||||
supportsVision: z.boolean().optional(),
|
||||
supportsWebSearch: z.boolean().optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const PricingTier = z.object({ usd: z.number(), diem: z.number().optional() }).passthrough();
|
||||
|
||||
const ExtendedPricing = z
|
||||
.object({
|
||||
context_token_threshold: z.number(),
|
||||
input: PricingTier,
|
||||
output: PricingTier,
|
||||
cache_input: PricingTier.optional(),
|
||||
cache_write: PricingTier.optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const Pricing = z
|
||||
.object({
|
||||
input: PricingTier,
|
||||
output: PricingTier,
|
||||
cache_input: PricingTier.optional(),
|
||||
cache_write: PricingTier.optional(),
|
||||
extended: ExtendedPricing.optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const ModelSpec = z
|
||||
.object({
|
||||
pricing: Pricing.optional(),
|
||||
availableContextTokens: z.number(),
|
||||
maxCompletionTokens: z.number().optional(),
|
||||
capabilities: Capabilities,
|
||||
constraints: z.any().optional(),
|
||||
name: z.string(),
|
||||
modelSource: z.string().optional(),
|
||||
offline: z.boolean().optional(),
|
||||
privacy: z.string().optional(),
|
||||
traits: z.array(z.string()).optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const VeniceModel = z
|
||||
.object({
|
||||
created: z.number(),
|
||||
id: z.string(),
|
||||
model_spec: ModelSpec,
|
||||
object: z.string(),
|
||||
owned_by: z.string(),
|
||||
type: z.string(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const VeniceResponse = z
|
||||
.object({
|
||||
data: z.array(VeniceModel),
|
||||
object: z.string(),
|
||||
type: z.string(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
function matchesFamily(target: string, family: string): boolean {
|
||||
const targetLower = target.toLowerCase();
|
||||
const familyLower = family.toLowerCase();
|
||||
let familyIdx = 0;
|
||||
|
||||
for (let i = 0; i < targetLower.length && familyIdx < familyLower.length; i++) {
|
||||
if (targetLower[i] === familyLower[familyIdx]) {
|
||||
familyIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
return familyIdx === familyLower.length;
|
||||
}
|
||||
|
||||
function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
const sortedFamilies = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
if (matchesFamily(modelId, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
if (matchesFamily(modelName, family)) {
|
||||
return family;
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function buildInputModalities(capabilities: z.infer<typeof Capabilities>): string[] {
|
||||
const mods: string[] = ["text"];
|
||||
if (capabilities.supportsVision) mods.push("image");
|
||||
if (capabilities.supportsAudioInput) mods.push("audio");
|
||||
if (capabilities.supportsVideoInput) mods.push("video");
|
||||
return mods;
|
||||
}
|
||||
|
||||
function formatNumber(n: number): string {
|
||||
if (n >= 1000) {
|
||||
// Format with underscores for readability (e.g., 131_072)
|
||||
return n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_");
|
||||
}
|
||||
return n.toString();
|
||||
}
|
||||
|
||||
function timestampToDate(timestamp: number): string {
|
||||
const date = new Date(timestamp * 1000);
|
||||
return date.toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function getTodayDate(): string {
|
||||
return new Date().toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
interface ExistingModel {
|
||||
name?: string;
|
||||
family?: string;
|
||||
attachment?: boolean;
|
||||
reasoning?: boolean;
|
||||
reasoning_options?: ReasoningOption[];
|
||||
tool_call?: boolean;
|
||||
structured_output?: boolean;
|
||||
temperature?: boolean;
|
||||
knowledge?: string;
|
||||
release_date?: string;
|
||||
last_updated?: string;
|
||||
open_weights?: boolean;
|
||||
interleaved?: boolean | { field: string };
|
||||
status?: string;
|
||||
cost?: {
|
||||
input?: number;
|
||||
output?: number;
|
||||
reasoning?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_over_200k?: {
|
||||
input?: number;
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_min?: number;
|
||||
};
|
||||
tiers?: Array<{
|
||||
tier: {
|
||||
type?: "context";
|
||||
size: number;
|
||||
};
|
||||
input?: number;
|
||||
output?: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
}>;
|
||||
};
|
||||
limit?: {
|
||||
context?: number;
|
||||
input?: number;
|
||||
output?: number;
|
||||
};
|
||||
modalities?: {
|
||||
input?: string[];
|
||||
output?: string[];
|
||||
};
|
||||
provider?: {
|
||||
npm?: string;
|
||||
api?: string;
|
||||
};
|
||||
}
|
||||
|
||||
async function loadExistingModel(filePath: string): Promise<ExistingModel | null> {
|
||||
try {
|
||||
const file = Bun.file(filePath);
|
||||
if (!(await file.exists())) {
|
||||
return null;
|
||||
}
|
||||
const toml = await import(filePath, { with: { type: "toml" } }).then(
|
||||
(mod) => mod.default,
|
||||
);
|
||||
return toml as ExistingModel;
|
||||
} catch (e) {
|
||||
console.warn(`Warning: Failed to parse existing file ${filePath}:`, e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function getExistingLongContextMin(existing: ExistingModel | null) {
|
||||
return (
|
||||
existing?.cost?.tiers?.find(
|
||||
(tier) =>
|
||||
(tier.tier.type === undefined || tier.tier.type === "context") &&
|
||||
tier.tier.size >= 200_000,
|
||||
)?.tier.size ?? 200_000
|
||||
);
|
||||
}
|
||||
|
||||
function getExistingLongContextCost(existing: ExistingModel | null) {
|
||||
return (
|
||||
existing?.cost?.tiers?.find(
|
||||
(tier) =>
|
||||
(tier.tier.type === undefined || tier.tier.type === "context") &&
|
||||
tier.tier.size >= 200_000,
|
||||
) ?? existing?.cost?.context_over_200k
|
||||
);
|
||||
}
|
||||
|
||||
function getLongContextMin(cost: { context_min?: number }) {
|
||||
return cost.context_min ?? 200_000;
|
||||
}
|
||||
|
||||
interface MergedModel {
|
||||
name: string;
|
||||
family?: string;
|
||||
attachment: boolean;
|
||||
reasoning: boolean;
|
||||
reasoning_options?: ReasoningOption[];
|
||||
tool_call: boolean;
|
||||
structured_output?: boolean;
|
||||
temperature: boolean;
|
||||
knowledge?: string;
|
||||
release_date: string;
|
||||
last_updated: string;
|
||||
open_weights: boolean;
|
||||
interleaved?: boolean | { field: string };
|
||||
status?: string;
|
||||
cost?: {
|
||||
input: number;
|
||||
output: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_over_200k?: {
|
||||
input: number;
|
||||
output: number;
|
||||
cache_read?: number;
|
||||
cache_write?: number;
|
||||
context_min?: number;
|
||||
};
|
||||
};
|
||||
limit: {
|
||||
context: number;
|
||||
output: number;
|
||||
};
|
||||
modalities: {
|
||||
input: string[];
|
||||
output: string[];
|
||||
};
|
||||
}
|
||||
|
||||
export function mergeModel(
|
||||
apiModel: z.infer<typeof VeniceModel>,
|
||||
existing: ExistingModel | null,
|
||||
reportDiscrepancy: (message: string) => void = console.warn,
|
||||
): MergedModel {
|
||||
const spec = apiModel.model_spec;
|
||||
const caps = spec.capabilities;
|
||||
|
||||
const contextTokens = spec.availableContextTokens;
|
||||
const outputTokens = spec.maxCompletionTokens ?? Math.floor(contextTokens / 4);
|
||||
|
||||
const openWeights = spec.modelSource?.toLowerCase().includes("huggingface") ?? false;
|
||||
|
||||
const inputModalities = buildInputModalities(caps);
|
||||
|
||||
if (existing?.modalities?.input?.includes("pdf") && !inputModalities.includes("pdf")) {
|
||||
inputModalities.push("pdf");
|
||||
}
|
||||
|
||||
const attachment =
|
||||
caps.supportsVision === true ||
|
||||
caps.supportsAudioInput === true ||
|
||||
caps.supportsVideoInput === true;
|
||||
|
||||
const merged: MergedModel = {
|
||||
name: spec.name,
|
||||
attachment,
|
||||
reasoning: caps.supportsReasoning === true,
|
||||
tool_call: caps.supportsFunctionCalling === true,
|
||||
temperature: true,
|
||||
release_date: timestampToDate(apiModel.created),
|
||||
last_updated: getTodayDate(),
|
||||
open_weights: openWeights,
|
||||
limit: {
|
||||
context: contextTokens,
|
||||
output: outputTokens,
|
||||
},
|
||||
modalities: {
|
||||
input: inputModalities,
|
||||
output: ["text"],
|
||||
},
|
||||
};
|
||||
|
||||
const override = REASONING_OVERRIDES[apiModel.id];
|
||||
const catalogOptions = caps.supportsReasoningEffort === true && caps.reasoningEffortOptions !== undefined
|
||||
? effort(...caps.reasoningEffortOptions)
|
||||
: undefined;
|
||||
const curatedOptions = existing?.reasoning_options;
|
||||
const selectedOptions = override ?? curatedOptions ?? (existing === null ? catalogOptions : undefined);
|
||||
|
||||
if (selectedOptions !== undefined) {
|
||||
merged.reasoning_options = selectedOptions;
|
||||
}
|
||||
|
||||
const catalogClaim = catalogOptions ?? (caps.supportsReasoningEffort === false ? [] : undefined);
|
||||
if (override !== undefined && JSON.stringify(override) !== JSON.stringify(curatedOptions) && curatedOptions !== undefined) {
|
||||
reportDiscrepancy(`${apiModel.id}: documented override replaces curated reasoning_options`);
|
||||
}
|
||||
if (
|
||||
selectedOptions !== undefined &&
|
||||
catalogClaim !== undefined &&
|
||||
JSON.stringify(selectedOptions) !== JSON.stringify(catalogClaim)
|
||||
) {
|
||||
reportDiscrepancy(
|
||||
`${apiModel.id}: preserving ${override !== undefined ? "documented override" : "curated reasoning_options"} despite catalog ${caps.supportsReasoningEffort === false ? "supportsReasoningEffort=false" : "option mismatch"}`,
|
||||
);
|
||||
}
|
||||
|
||||
// structured_output only if true
|
||||
if (caps.supportsResponseSchema === true) {
|
||||
merged.structured_output = true;
|
||||
}
|
||||
|
||||
// Cost from API
|
||||
if (spec.pricing) {
|
||||
merged.cost = {
|
||||
input: spec.pricing.input.usd,
|
||||
output: spec.pricing.output.usd,
|
||||
...(spec.pricing.cache_input && { cache_read: spec.pricing.cache_input.usd }),
|
||||
...(spec.pricing.cache_write && { cache_write: spec.pricing.cache_write.usd }),
|
||||
};
|
||||
|
||||
// Extended pricing maps to context_over_200k
|
||||
if (spec.pricing.extended) {
|
||||
merged.cost.context_over_200k = {
|
||||
input: spec.pricing.extended.input.usd,
|
||||
output: spec.pricing.extended.output.usd,
|
||||
context_min: spec.pricing.extended.context_token_threshold,
|
||||
...(spec.pricing.extended.cache_input && { cache_read: spec.pricing.extended.cache_input.usd }),
|
||||
...(spec.pricing.extended.cache_write && { cache_write: spec.pricing.extended.cache_write.usd }),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
const inferred = inferFamily(apiModel.id, spec.name);
|
||||
merged.family = inferred ?? existing?.family;
|
||||
|
||||
// Preserve manual fields from existing
|
||||
if (existing?.knowledge) {
|
||||
merged.knowledge = existing.knowledge;
|
||||
}
|
||||
if (existing?.interleaved !== undefined) {
|
||||
merged.interleaved = existing.interleaved;
|
||||
}
|
||||
if (existing?.status !== undefined) {
|
||||
merged.status = existing.status;
|
||||
}
|
||||
|
||||
return merged;
|
||||
}
|
||||
|
||||
export function formatToml(model: MergedModel): string {
|
||||
const lines: string[] = [];
|
||||
|
||||
// Basic fields
|
||||
lines.push(`name = "${model.name.replace(/"/g, '\\"')}"`);
|
||||
if (model.family) {
|
||||
lines.push(`family = "${model.family}"`);
|
||||
}
|
||||
lines.push(`attachment = ${model.attachment}`);
|
||||
lines.push(`reasoning = ${model.reasoning}`);
|
||||
if (model.reasoning_options?.length === 0) {
|
||||
lines.push("reasoning_options = []");
|
||||
}
|
||||
lines.push(`tool_call = ${model.tool_call}`);
|
||||
if (model.structured_output !== undefined) {
|
||||
lines.push(`structured_output = ${model.structured_output}`);
|
||||
}
|
||||
lines.push(`temperature = ${model.temperature}`);
|
||||
if (model.knowledge) {
|
||||
lines.push(`knowledge = "${model.knowledge}"`);
|
||||
}
|
||||
lines.push(`release_date = "${model.release_date}"`);
|
||||
lines.push(`last_updated = "${model.last_updated}"`);
|
||||
lines.push(`open_weights = ${model.open_weights}`);
|
||||
if (model.status) {
|
||||
lines.push(`status = "${model.status}"`);
|
||||
}
|
||||
|
||||
for (const option of model.reasoning_options ?? []) {
|
||||
lines.push("");
|
||||
lines.push("[[reasoning_options]]");
|
||||
lines.push(`type = "${option.type}"`);
|
||||
lines.push(`values = [${option.values.map((value) => `"${value}"`).join(", ")}]`);
|
||||
}
|
||||
|
||||
// Interleaved section (if present)
|
||||
if (model.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (model.interleaved === true) {
|
||||
lines.push(`interleaved = true`);
|
||||
} else if (typeof model.interleaved === "object") {
|
||||
lines.push(`[interleaved]`);
|
||||
lines.push(`field = "${model.interleaved.field}"`);
|
||||
}
|
||||
}
|
||||
|
||||
// Cost section
|
||||
if (model.cost) {
|
||||
lines.push("");
|
||||
lines.push(`[cost]`);
|
||||
lines.push(`input = ${model.cost.input}`);
|
||||
lines.push(`output = ${model.cost.output}`);
|
||||
if (model.cost.cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${model.cost.cache_read}`);
|
||||
}
|
||||
if (model.cost.cache_write !== undefined) {
|
||||
lines.push(`cache_write = ${model.cost.cache_write}`);
|
||||
}
|
||||
|
||||
if (model.cost.context_over_200k) {
|
||||
lines.push("");
|
||||
lines.push(`[[cost.tiers]]`);
|
||||
lines.push(`tier = { size = ${formatNumber(getLongContextMin(model.cost.context_over_200k))} }`);
|
||||
lines.push(`input = ${model.cost.context_over_200k.input}`);
|
||||
lines.push(`output = ${model.cost.context_over_200k.output}`);
|
||||
if (model.cost.context_over_200k.cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${model.cost.context_over_200k.cache_read}`);
|
||||
}
|
||||
if (model.cost.context_over_200k.cache_write !== undefined) {
|
||||
lines.push(`cache_write = ${model.cost.context_over_200k.cache_write}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Limit section
|
||||
lines.push("");
|
||||
lines.push(`[limit]`);
|
||||
lines.push(`context = ${formatNumber(model.limit.context)}`);
|
||||
lines.push(`output = ${formatNumber(model.limit.output)}`);
|
||||
|
||||
// Modalities section
|
||||
lines.push("");
|
||||
lines.push(`[modalities]`);
|
||||
lines.push(`input = [${model.modalities.input.map((m) => `"${m}"`).join(", ")}]`);
|
||||
lines.push(`output = [${model.modalities.output.map((m) => `"${m}"`).join(", ")}]`);
|
||||
|
||||
return lines.join("\n") + "\n";
|
||||
}
|
||||
|
||||
interface Changes {
|
||||
field: string;
|
||||
oldValue: string;
|
||||
newValue: string;
|
||||
}
|
||||
|
||||
export function detectChanges(
|
||||
existing: ExistingModel | null,
|
||||
merged: MergedModel,
|
||||
): Changes[] {
|
||||
if (!existing) return [];
|
||||
|
||||
const changes: Changes[] = [];
|
||||
|
||||
const compare = (field: string, oldVal: unknown, newVal: unknown) => {
|
||||
const oldStr = JSON.stringify(oldVal);
|
||||
const newStr = JSON.stringify(newVal);
|
||||
if (oldStr !== newStr) {
|
||||
changes.push({
|
||||
field,
|
||||
oldValue: formatValue(oldVal),
|
||||
newValue: formatValue(newVal),
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
const formatValue = (val: unknown): string => {
|
||||
if (typeof val === "number") return formatNumber(val);
|
||||
if (Array.isArray(val)) return JSON.stringify(val);
|
||||
if (val === undefined) return "(none)";
|
||||
return String(val);
|
||||
};
|
||||
|
||||
compare("name", existing.name, merged.name);
|
||||
compare("family", existing.family, merged.family);
|
||||
compare("attachment", existing.attachment, merged.attachment);
|
||||
compare("reasoning", existing.reasoning, merged.reasoning);
|
||||
compare("reasoning_options", existing.reasoning_options, merged.reasoning_options);
|
||||
compare("tool_call", existing.tool_call, merged.tool_call);
|
||||
compare("structured_output", existing.structured_output, merged.structured_output);
|
||||
compare("open_weights", existing.open_weights, merged.open_weights);
|
||||
compare("release_date", existing.release_date, merged.release_date);
|
||||
compare("cost.input", existing.cost?.input, merged.cost?.input);
|
||||
compare("cost.output", existing.cost?.output, merged.cost?.output);
|
||||
compare("cost.cache_read", existing.cost?.cache_read, merged.cost?.cache_read);
|
||||
compare("cost.cache_write", existing.cost?.cache_write, merged.cost?.cache_write);
|
||||
const existingLongContextCost = getExistingLongContextCost(existing);
|
||||
compare("cost.context_over_200k.input", existingLongContextCost?.input, merged.cost?.context_over_200k?.input);
|
||||
compare("cost.context_over_200k.output", existingLongContextCost?.output, merged.cost?.context_over_200k?.output);
|
||||
compare("cost.context_over_200k.cache_read", existingLongContextCost?.cache_read, merged.cost?.context_over_200k?.cache_read);
|
||||
compare("cost.context_over_200k.cache_write", existingLongContextCost?.cache_write, merged.cost?.context_over_200k?.cache_write);
|
||||
compare("limit.context", existing.limit?.context, merged.limit.context);
|
||||
compare("limit.output", existing.limit?.output, merged.limit.output);
|
||||
compare("modalities.input", existing.modalities?.input, merged.modalities.input);
|
||||
|
||||
return changes;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const args = process.argv.slice(2);
|
||||
const dryRun = args.includes("--dry-run");
|
||||
|
||||
const modelsDir = path.join(
|
||||
import.meta.dirname,
|
||||
"..",
|
||||
"..",
|
||||
"..",
|
||||
"providers",
|
||||
"venice",
|
||||
"models",
|
||||
);
|
||||
|
||||
// Check for API key from CLI argument or environment variable
|
||||
let apiKey: string | null = null;
|
||||
|
||||
// Check CLI args for --api-key=xxx or --api-key xxx
|
||||
const apiKeyArgIndex = args.findIndex((arg) => arg.startsWith("--api-key"));
|
||||
if (apiKeyArgIndex !== -1) {
|
||||
const arg = args[apiKeyArgIndex];
|
||||
if (arg?.includes("=")) {
|
||||
apiKey = arg.split("=")[1] ?? null;
|
||||
} else if (args[apiKeyArgIndex + 1]) {
|
||||
apiKey = args[apiKeyArgIndex + 1] ?? null;
|
||||
}
|
||||
}
|
||||
|
||||
// Fall back to environment variable
|
||||
if (!apiKey) {
|
||||
apiKey = process.env.VENICE_API_KEY ?? null;
|
||||
}
|
||||
|
||||
const includeAlpha = apiKey !== null;
|
||||
|
||||
if (dryRun) {
|
||||
console.log(
|
||||
`[DRY RUN] Fetching Venice models from API${includeAlpha ? " (including alpha models)" : ""}...`,
|
||||
);
|
||||
} else {
|
||||
console.log(
|
||||
`Fetching Venice models from API${includeAlpha ? " (including alpha models)" : ""}...`,
|
||||
);
|
||||
}
|
||||
|
||||
// Fetch API data
|
||||
const fetchOptions: RequestInit = {};
|
||||
if (apiKey) {
|
||||
fetchOptions.headers = {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
};
|
||||
}
|
||||
|
||||
const res = await fetch(API_ENDPOINT, fetchOptions);
|
||||
if (!res.ok) {
|
||||
console.error(`Failed to fetch API: ${res.status} ${res.statusText}`);
|
||||
if (res.status === 401) {
|
||||
console.error("Invalid API key. Please check your VENICE_API_KEY.");
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const json = await res.json();
|
||||
const parsed = VeniceResponse.safeParse(json);
|
||||
if (!parsed.success) {
|
||||
console.error("Invalid API response:", parsed.error.errors);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const apiModels = parsed.data.data;
|
||||
|
||||
// Get existing files
|
||||
const existingFiles = new Set<string>();
|
||||
try {
|
||||
const files = await readdir(modelsDir);
|
||||
for (const file of files) {
|
||||
if (file.endsWith(".toml")) {
|
||||
existingFiles.add(file);
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Directory might not exist yet
|
||||
}
|
||||
|
||||
console.log(`Found ${apiModels.length} models in API, ${existingFiles.size} existing files\n`);
|
||||
|
||||
// Track API model IDs for orphan detection
|
||||
const apiModelIds = new Set<string>();
|
||||
|
||||
let created = 0;
|
||||
let updated = 0;
|
||||
let unchanged = 0;
|
||||
|
||||
for (const apiModel of apiModels) {
|
||||
const safeId = apiModel.id.replace(/\//g, "-");
|
||||
const filename = `${safeId}.toml`;
|
||||
const filePath = path.join(modelsDir, filename);
|
||||
|
||||
apiModelIds.add(filename);
|
||||
|
||||
const existing = await loadExistingModel(filePath);
|
||||
const merged = mergeModel(apiModel, existing);
|
||||
const tomlContent = formatToml(merged);
|
||||
|
||||
if (existing === null) {
|
||||
// New file
|
||||
created++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would create: ${filename}`);
|
||||
console.log(` name = "${merged.name}"`);
|
||||
if (merged.family) {
|
||||
console.log(` family = "${merged.family}" (inferred)`);
|
||||
}
|
||||
console.log("");
|
||||
} else {
|
||||
await Bun.write(filePath, tomlContent);
|
||||
console.log(`Created: ${filename}`);
|
||||
}
|
||||
} else {
|
||||
// Check for changes
|
||||
const changes = detectChanges(existing, merged);
|
||||
|
||||
if (changes.length > 0) {
|
||||
updated++;
|
||||
if (dryRun) {
|
||||
console.log(`[DRY RUN] Would update: ${filename}`);
|
||||
} else {
|
||||
await Bun.write(filePath, tomlContent);
|
||||
console.log(`Updated: ${filename}`);
|
||||
}
|
||||
for (const change of changes) {
|
||||
console.log(` ${change.field}: ${change.oldValue} → ${change.newValue}`);
|
||||
}
|
||||
console.log("");
|
||||
} else {
|
||||
unchanged++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check for orphaned files
|
||||
const orphaned: string[] = [];
|
||||
for (const file of existingFiles) {
|
||||
if (!apiModelIds.has(file)) {
|
||||
orphaned.push(file);
|
||||
console.log(`Warning: Orphaned file (not in API): ${file}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Summary
|
||||
console.log("");
|
||||
if (dryRun) {
|
||||
console.log(
|
||||
`Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} orphaned`,
|
||||
);
|
||||
} else {
|
||||
console.log(
|
||||
`Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.main) {
|
||||
await main();
|
||||
}
|
||||
@@ -3,7 +3,7 @@
|
||||
import path from "node:path";
|
||||
import { mkdir } from "node:fs/promises";
|
||||
import { z } from "zod";
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../src/family.js";
|
||||
import { ModelFamilyValues } from "../src/family.js";
|
||||
|
||||
const API_ENDPOINT = "https://trace.wandb.ai/inference/analysis/artificialanalysis/models";
|
||||
|
||||
@@ -176,9 +176,6 @@ function matchesFamily(target: string, family: string): boolean {
|
||||
}
|
||||
|
||||
function inferFamily(modelId: string, modelName: string): string | undefined {
|
||||
const kimiFamily = inferKimiFamily(modelId, modelName);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const sortedFamilies = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const family of sortedFamilies) {
|
||||
|
||||
@@ -26,7 +26,6 @@ export const ModelFamilyValues = [
|
||||
"claude-haiku",
|
||||
"claude-sonnet",
|
||||
"claude-opus",
|
||||
"claude-fable",
|
||||
|
||||
// Gemini style
|
||||
"gemini",
|
||||
@@ -66,7 +65,8 @@ export const ModelFamilyValues = [
|
||||
|
||||
// Moonshot Kimi
|
||||
"kimi",
|
||||
"kimi-k2",
|
||||
"kimi-k2.5",
|
||||
"kimi-k2.6",
|
||||
"kimi-free",
|
||||
"kimi-thinking",
|
||||
|
||||
@@ -102,8 +102,6 @@ export const ModelFamilyValues = [
|
||||
"command-r",
|
||||
"command-a",
|
||||
"command-light",
|
||||
"north",
|
||||
"north-free",
|
||||
|
||||
// AI21 Jamba
|
||||
"jamba",
|
||||
@@ -246,9 +244,6 @@ export const ModelFamilyValues = [
|
||||
// Lucid
|
||||
"lucid",
|
||||
|
||||
// LucidQuery
|
||||
"agi",
|
||||
|
||||
// Intellect
|
||||
"intellect",
|
||||
|
||||
@@ -425,10 +420,3 @@ export const ModelFamilyValues = [
|
||||
|
||||
export const ModelFamily = z.enum(ModelFamilyValues);
|
||||
export type ModelFamily = z.infer<typeof ModelFamily>;
|
||||
|
||||
export function inferKimiFamily(...values: string[]): ModelFamily | undefined {
|
||||
const target = values.join(" ").toLowerCase();
|
||||
if (/kimi[^a-z0-9]*k2(?:[^a-z0-9]*\d+)?[^a-z0-9]*thinking/.test(target)) return "kimi-thinking";
|
||||
if (/kimi[\s_-]*k2/.test(target)) return "kimi-k2";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
@@ -39,7 +39,7 @@ export async function generateModels(directory: string) {
|
||||
absolute: true,
|
||||
followSymlinks: true,
|
||||
})) {
|
||||
const modelID = path.relative(directory, modelPath).split(path.sep).join("/").slice(0, -5);
|
||||
const modelID = path.relative(directory, modelPath).slice(0, -5);
|
||||
const toml = await import(modelPath, {
|
||||
with: {
|
||||
type: "toml",
|
||||
@@ -94,7 +94,7 @@ async function generateProviders(
|
||||
absolute: true,
|
||||
followSymlinks: true,
|
||||
})) {
|
||||
const modelID = path.relative(modelsPath, modelPath).split(path.sep).join("/").slice(0, -5);
|
||||
const modelID = path.relative(modelsPath, modelPath).slice(0, -5);
|
||||
const toml = await import(modelPath, {
|
||||
with: {
|
||||
type: "toml",
|
||||
|
||||
@@ -25,7 +25,7 @@ const ReasoningEffortValue = z.preprocess(
|
||||
(value) => (value === "null" ? null : value),
|
||||
z.union([
|
||||
z.null(),
|
||||
z.enum(["none", "minimal", "low", "medium", "high", "xhigh", "max", "default"]),
|
||||
z.enum(["none", "minimal", "low", "medium", "high", "xhigh", "max"]),
|
||||
]),
|
||||
);
|
||||
|
||||
@@ -47,7 +47,7 @@ const ReasoningOption = z
|
||||
type: z.literal("budget_tokens"),
|
||||
min: z
|
||||
.number()
|
||||
.min(-1, "Minimum reasoning budget cannot be less than -1")
|
||||
.min(0, "Minimum reasoning budget cannot be negative")
|
||||
.optional(),
|
||||
max: z
|
||||
.number()
|
||||
|
||||
+23
-156
@@ -3,17 +3,12 @@ import { lstat, mkdir, readdir, rm } from "node:fs/promises";
|
||||
import { mergeDeep } from "remeda";
|
||||
import { z } from "zod";
|
||||
|
||||
import { AuthoredModel, AuthoredModelShape, ModelMetadata } from "../schema.js";
|
||||
import { baseten } from "./providers/baseten.js";
|
||||
import { chutes } from "./providers/chutes.js";
|
||||
import { AuthoredModel, AuthoredModelShape } from "../schema.js";
|
||||
import { cloudflareWorkersAi } from "./providers/cloudflare-workers-ai.js";
|
||||
import { google } from "./providers/google.js";
|
||||
import { huggingface } from "./providers/huggingface.js";
|
||||
import { llmgateway } from "./providers/llmgateway.js";
|
||||
import { openrouter } from "./providers/openrouter.js";
|
||||
import { ovhcloud } from "./providers/ovhcloud.js";
|
||||
import { vercel } from "./providers/vercel.js";
|
||||
import { venice } from "./providers/venice.js";
|
||||
import { xai } from "./providers/xai.js";
|
||||
|
||||
const ExistingModelType = AuthoredModelShape.partial()
|
||||
@@ -44,17 +39,14 @@ export type ExistingModel = z.infer<typeof ExistingModelType>;
|
||||
export type SyncedFullModel = Omit<z.infer<typeof AuthoredModelShape>, "id">;
|
||||
export type SyncedBaseModel = Omit<z.infer<typeof SyncedBaseModel>, "id">;
|
||||
export type SyncedModel = SyncedFullModel | SyncedBaseModel;
|
||||
export type SyncedMetadata = Omit<z.infer<typeof ModelMetadata>, "id">;
|
||||
|
||||
export interface SyncProvider<SourceModel> {
|
||||
id: string;
|
||||
name: string;
|
||||
modelsDir: string;
|
||||
metadataNamespace?: string;
|
||||
skipCreates?: boolean;
|
||||
deleteMissing?: boolean;
|
||||
preserveSymlinks?: boolean;
|
||||
preserveBaseModels?: boolean;
|
||||
sameModel?(current: ExistingModel, desired: SyncedModel): boolean;
|
||||
missingNotice?(paths: string[]): string[];
|
||||
sourceID?(model: SourceModel): string;
|
||||
@@ -64,7 +56,7 @@ export interface SyncProvider<SourceModel> {
|
||||
translateModel(
|
||||
model: SourceModel,
|
||||
context: { existing(id: string): ExistingModel | undefined },
|
||||
): { id: string; model: SyncedModel; metadata?: { id: string; model: SyncedMetadata } } | undefined;
|
||||
): { id: string; model: SyncedModel } | undefined;
|
||||
}
|
||||
|
||||
export interface SyncResult {
|
||||
@@ -80,35 +72,25 @@ export interface SyncResult {
|
||||
}
|
||||
|
||||
export const providers: {
|
||||
baseten: SyncProvider<any>;
|
||||
chutes: SyncProvider<any>;
|
||||
"cloudflare-workers-ai": SyncProvider<any>;
|
||||
google: SyncProvider<any>;
|
||||
huggingface: SyncProvider<any>;
|
||||
llmgateway: SyncProvider<any>;
|
||||
openrouter: SyncProvider<any>;
|
||||
ovhcloud: SyncProvider<any>;
|
||||
vercel: SyncProvider<any>;
|
||||
venice: SyncProvider<any>;
|
||||
xai: SyncProvider<any>;
|
||||
} = {
|
||||
baseten,
|
||||
chutes,
|
||||
"cloudflare-workers-ai": cloudflareWorkersAi,
|
||||
google,
|
||||
huggingface,
|
||||
llmgateway,
|
||||
openrouter,
|
||||
ovhcloud,
|
||||
vercel,
|
||||
venice,
|
||||
xai,
|
||||
};
|
||||
|
||||
export const groups = {
|
||||
aggregators: ["huggingface", "llmgateway", "openrouter", "vercel"],
|
||||
aggregators: ["openrouter", "vercel"],
|
||||
cloudflare: ["cloudflare-workers-ai"],
|
||||
direct: ["baseten", "chutes", "google", "ovhcloud", "venice", "xai"],
|
||||
direct: ["google", "ovhcloud", "xai"],
|
||||
} as const;
|
||||
|
||||
type ProviderID = keyof typeof providers;
|
||||
@@ -128,12 +110,9 @@ export async function syncProvider<SourceModel>(
|
||||
): Promise<SyncResult> {
|
||||
console.log(`\nSyncing ${provider.name}...`);
|
||||
|
||||
const existingState = await readExisting(provider.modelsDir);
|
||||
const { models: existing, brokenSymlinks } = existingState;
|
||||
let { modelMetadata } = existingState;
|
||||
const { models: existing, brokenSymlinks } = await readExisting(provider.modelsDir);
|
||||
const sourceModels = provider.parseModels(await provider.fetchModels());
|
||||
const desired = new Map<string, { model: z.infer<typeof SyncedAuthoredModel>; content: string }>();
|
||||
const desiredMetadata = new Map<string, { model: z.infer<typeof ModelMetadata>; content: string }>();
|
||||
const skippedRemote: string[] = [];
|
||||
|
||||
for (const sourceModel of sourceModels) {
|
||||
@@ -143,7 +122,7 @@ export async function syncProvider<SourceModel>(
|
||||
},
|
||||
});
|
||||
if (translated === undefined) {
|
||||
if (provider.sourceID !== undefined) skippedRemote.push(provider.sourceID(sourceModel));
|
||||
if (provider.skipCreates) skippedRemote.push(provider.sourceID?.(sourceModel) ?? "unknown");
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -157,46 +136,9 @@ export async function syncProvider<SourceModel>(
|
||||
throw new Error(`Duplicate synced model path: ${provider.id}/${relativePath}`);
|
||||
}
|
||||
|
||||
if (translated.metadata !== undefined) {
|
||||
const parsedMetadata = ModelMetadata.safeParse({
|
||||
id: translated.metadata.id,
|
||||
...stripUndefined(translated.metadata.model),
|
||||
});
|
||||
if (!parsedMetadata.success) {
|
||||
parsedMetadata.error.cause = { provider: provider.id, metadata: translated.metadata.id };
|
||||
throw parsedMetadata.error;
|
||||
}
|
||||
const metadataPath = `${translated.metadata.id}.toml`;
|
||||
if (desiredMetadata.has(metadataPath)) throw new Error(`Duplicate synced metadata path: ${metadataPath}`);
|
||||
desiredMetadata.set(metadataPath, {
|
||||
model: parsedMetadata.data,
|
||||
content: formatMetadataToml(parsedMetadata.data),
|
||||
});
|
||||
}
|
||||
|
||||
const translatedModel = provider.preserveBaseModels === false
|
||||
? translated.model
|
||||
: preserveBaseModel(translated.model, existing.get(relativePath)?.authored);
|
||||
const translatedBase = "base_model" in translatedModel ? translatedModel.base_model : undefined;
|
||||
let resolvedReasoning: boolean | undefined;
|
||||
if (translatedBase !== undefined) {
|
||||
if (translated.metadata?.id === translatedBase) {
|
||||
resolvedReasoning = translated.metadata.model.reasoning;
|
||||
} else {
|
||||
modelMetadata ??= await readModelMetadata(provider.modelsDir);
|
||||
const canonicalReasoning = modelMetadata[translatedBase]?.reasoning;
|
||||
resolvedReasoning = typeof canonicalReasoning === "boolean" ? canonicalReasoning : undefined;
|
||||
}
|
||||
} else {
|
||||
resolvedReasoning = existing.get(relativePath)?.toml.reasoning;
|
||||
}
|
||||
const parsed = SyncedAuthoredModel.safeParse(stripUndefined({
|
||||
id: translated.id,
|
||||
...preserveReasoningOptions(
|
||||
translatedModel,
|
||||
existing.get(relativePath)?.authored,
|
||||
resolvedReasoning,
|
||||
),
|
||||
...preserveBaseModel(translated.model, existing.get(relativePath)?.authored),
|
||||
}));
|
||||
if (!parsed.success) {
|
||||
parsed.error.cause = { provider: provider.id, path: relativePath };
|
||||
@@ -212,48 +154,6 @@ export async function syncProvider<SourceModel>(
|
||||
const files: SyncResult["files"] = [];
|
||||
let unchanged = 0;
|
||||
|
||||
const metadataDir = modelMetadataDir(provider.modelsDir);
|
||||
for (const [relativePath, file] of desiredMetadata) {
|
||||
const filePath = path.join(metadataDir, relativePath);
|
||||
const currentFile = Bun.file(filePath);
|
||||
const current = await currentFile.exists()
|
||||
? ModelMetadata.safeParse({
|
||||
id: relativePath.slice(0, -5),
|
||||
...Bun.TOML.parse(await currentFile.text()) as Record<string, unknown>,
|
||||
})
|
||||
: undefined;
|
||||
if (current?.success && stable(current.data) === stable(file.model)) continue;
|
||||
files.push({ status: current === undefined ? "created" : "updated", path: filePath });
|
||||
if (options.dryRun) {
|
||||
console.log(`Would ${current === undefined ? "create" : "update"} metadata ${relativePath}`);
|
||||
} else {
|
||||
await mkdir(path.dirname(filePath), { recursive: true });
|
||||
await Bun.write(filePath, file.content);
|
||||
}
|
||||
}
|
||||
|
||||
if (provider.metadataNamespace !== undefined) {
|
||||
if (!/^[a-z0-9-]+$/.test(provider.metadataNamespace)) {
|
||||
throw new Error(`Invalid metadata namespace: ${provider.metadataNamespace}`);
|
||||
}
|
||||
const namespaceDir = path.join(metadataDir, provider.metadataNamespace);
|
||||
for (const { file } of await tomlFiles(namespaceDir)) {
|
||||
const relativePath = path.join(provider.metadataNamespace, file).split(path.sep).join("/");
|
||||
if (desiredMetadata.has(relativePath) || provider.deleteMissing === false) continue;
|
||||
if (options.newOnly) {
|
||||
console.log(`Skipping metadata removal in new-only mode: ${relativePath}`);
|
||||
continue;
|
||||
}
|
||||
const filePath = path.join(metadataDir, relativePath);
|
||||
files.push({ status: "deleted", path: filePath });
|
||||
if (options.dryRun) {
|
||||
console.log(`Would remove metadata ${relativePath}`);
|
||||
} else {
|
||||
await rm(filePath, { force: true });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const [relativePath, file] of desired) {
|
||||
const filePath = path.join(provider.modelsDir, relativePath);
|
||||
const current = existing.get(relativePath);
|
||||
@@ -342,22 +242,6 @@ export function preserveBaseModel(model: SyncedModel, existing: ExistingModel |
|
||||
};
|
||||
}
|
||||
|
||||
export function preserveReasoningOptions(
|
||||
model: SyncedModel,
|
||||
existing: ExistingModel | undefined,
|
||||
resolvedReasoning: boolean | undefined = existing?.reasoning,
|
||||
): SyncedModel {
|
||||
if ((model.reasoning ?? resolvedReasoning) === false) {
|
||||
const { reasoning_options: _reasoningOptions, ...withoutReasoningOptions } = model;
|
||||
return withoutReasoningOptions as SyncedModel;
|
||||
}
|
||||
if (model.reasoning_options !== undefined || existing?.reasoning_options === undefined) return model;
|
||||
return {
|
||||
...model,
|
||||
reasoning_options: existing.reasoning_options,
|
||||
};
|
||||
}
|
||||
|
||||
export async function syncTargets(target: string, options: SyncOptions = {}) {
|
||||
const ids = target in groups
|
||||
? groups[target as keyof typeof groups]
|
||||
@@ -423,7 +307,7 @@ async function readExisting(modelsDir: string) {
|
||||
existing.set(file, { authored, toml, symlink });
|
||||
}
|
||||
|
||||
return { models: existing, brokenSymlinks, modelMetadata };
|
||||
return { models: existing, brokenSymlinks };
|
||||
}
|
||||
|
||||
async function isSymlink(filePath: string) {
|
||||
@@ -436,7 +320,8 @@ async function isSymlink(filePath: string) {
|
||||
}
|
||||
|
||||
async function readModelMetadata(modelsDir: string) {
|
||||
const metadataDir = modelMetadataDir(modelsDir);
|
||||
const root = path.dirname(path.dirname(path.dirname(modelsDir)));
|
||||
const metadataDir = path.join(root, "models");
|
||||
const result: Record<string, Record<string, unknown>> = {};
|
||||
|
||||
for await (const modelPath of new Bun.Glob("**/*.toml").scan({
|
||||
@@ -444,7 +329,7 @@ async function readModelMetadata(modelsDir: string) {
|
||||
absolute: true,
|
||||
followSymlinks: true,
|
||||
})) {
|
||||
const modelID = path.relative(metadataDir, modelPath).split(path.sep).join("/").slice(0, -5);
|
||||
const modelID = path.relative(metadataDir, modelPath).slice(0, -5);
|
||||
const toml = Bun.TOML.parse(
|
||||
await Bun.file(modelPath).text(),
|
||||
) as Record<string, unknown>;
|
||||
@@ -454,10 +339,6 @@ async function readModelMetadata(modelsDir: string) {
|
||||
return result;
|
||||
}
|
||||
|
||||
function modelMetadataDir(modelsDir: string) {
|
||||
return path.join(path.dirname(path.dirname(path.dirname(modelsDir))), "models");
|
||||
}
|
||||
|
||||
function resolveBaseModel(
|
||||
authored: ExistingModel,
|
||||
modelMetadata: Record<string, Record<string, unknown>>,
|
||||
@@ -553,7 +434,7 @@ async function tomlFiles(root: string, dir = "") {
|
||||
const result: Array<{ file: string; symlink: boolean }> = [];
|
||||
|
||||
for (const entry of await readdir(path.join(root, dir), { withFileTypes: true })) {
|
||||
const file = path.join(dir, entry.name).split(path.sep).join("/");
|
||||
const file = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
result.push(...await tomlFiles(root, file));
|
||||
} else if (entry.name.endsWith(".toml") && (entry.isFile() || entry.isSymbolicLink())) {
|
||||
@@ -682,7 +563,7 @@ function formatReasoningValue(value: string | null) {
|
||||
return value === null ? quote("null") : quote(value);
|
||||
}
|
||||
|
||||
export function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
|
||||
function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
|
||||
const lines: string[] = [];
|
||||
|
||||
if (model.base_model !== undefined) lines.push(`base_model = ${quote(model.base_model)}`);
|
||||
@@ -703,17 +584,6 @@ export function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
|
||||
if (model.knowledge !== undefined) lines.push(`knowledge = ${quote(model.knowledge)}`);
|
||||
if (model.open_weights !== undefined) lines.push(`open_weights = ${model.open_weights}`);
|
||||
if (model.status !== undefined) lines.push(`status = ${quote(model.status)}`);
|
||||
if (model.reasoning_options?.length === 0) lines.push("reasoning_options = []");
|
||||
|
||||
if (model.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (model.interleaved === true) {
|
||||
lines.push("interleaved = true");
|
||||
} else {
|
||||
lines.push("[interleaved]");
|
||||
lines.push(`field = ${quote(model.interleaved.field)}`);
|
||||
}
|
||||
}
|
||||
|
||||
for (const option of model.reasoning_options ?? []) {
|
||||
lines.push("", "[[reasoning_options]]");
|
||||
@@ -727,6 +597,16 @@ export function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
|
||||
}
|
||||
}
|
||||
|
||||
if (model.interleaved !== undefined) {
|
||||
lines.push("");
|
||||
if (model.interleaved === true) {
|
||||
lines.push("interleaved = true");
|
||||
} else {
|
||||
lines.push("[interleaved]");
|
||||
lines.push(`field = ${quote(model.interleaved.field)}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (model.cost !== undefined) {
|
||||
lines.push("", "[cost]");
|
||||
lines.push(`input = ${formatNumber(model.cost.input)}`);
|
||||
@@ -778,19 +658,6 @@ export function formatToml(model: z.infer<typeof SyncedAuthoredModel>) {
|
||||
return `${lines.join("\n")}\n`;
|
||||
}
|
||||
|
||||
function formatMetadataToml(model: z.infer<typeof ModelMetadata>) {
|
||||
const content = formatToml(model as unknown as z.infer<typeof SyncedAuthoredModel>).trimEnd();
|
||||
const lines = [content];
|
||||
for (const weight of model.weights ?? []) {
|
||||
lines.push("", "[[weights]]");
|
||||
if (weight.label !== undefined) lines.push(`label = ${quote(weight.label)}`);
|
||||
lines.push(`url = ${quote(weight.url)}`);
|
||||
if (weight.format !== undefined) lines.push(`format = ${quote(weight.format)}`);
|
||||
if (weight.quantization !== undefined) lines.push(`quantization = ${quote(weight.quantization)}`);
|
||||
}
|
||||
return `${lines.join("\n")}\n`;
|
||||
}
|
||||
|
||||
export async function main(args = process.argv.slice(2)) {
|
||||
if (args.includes("--list-providers")) {
|
||||
console.log(JSON.stringify(syncProviderMatrix()));
|
||||
|
||||
@@ -1,188 +0,0 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://inference.baseten.co/v1/models";
|
||||
|
||||
const Price = z.union([z.string(), z.number()]);
|
||||
|
||||
export const BasetenModel = z.object({
|
||||
id: z.string().min(1),
|
||||
name: z.string().min(1),
|
||||
context_length: z.number().int().positive(),
|
||||
max_completion_tokens: z.number().int().positive(),
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
pricing: z.object({
|
||||
prompt: Price,
|
||||
completion: Price,
|
||||
}).passthrough(),
|
||||
supported_features: z.array(z.string()),
|
||||
supported_sampling_parameters: z.array(z.string()),
|
||||
}).passthrough();
|
||||
|
||||
export const BasetenResponse = z.object({
|
||||
data: z.array(BasetenModel),
|
||||
}).passthrough();
|
||||
|
||||
export type BasetenModel = z.infer<typeof BasetenModel>;
|
||||
|
||||
export const baseten = {
|
||||
id: "baseten",
|
||||
name: "Baseten",
|
||||
modelsDir: "providers/baseten/models",
|
||||
deleteMissing: false,
|
||||
sourceID(model) {
|
||||
return model.id;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} Baseten models were not created because their slugs could not be mapped exactly to provider-agnostic metadata.`,
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
missingNotice(paths) {
|
||||
if (paths.length === 0) return [];
|
||||
return [
|
||||
`${paths.length} local Baseten models were absent from the catalog and were retained for manual lifecycle review.`,
|
||||
`Retained local paths: ${paths.map((item) => `\`${item}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
async fetchModels() {
|
||||
const key = process.env.BASETEN_API_KEY;
|
||||
if (key === undefined) throw new Error("Baseten sync requires BASETEN_API_KEY");
|
||||
return fetchBasetenModels(key);
|
||||
},
|
||||
parseModels(raw) {
|
||||
return BasetenResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const existing = context.existing(model.id);
|
||||
const baseModel = existing === undefined
|
||||
? resolveBasetenBaseModel(model.id)
|
||||
: existing.base_model;
|
||||
if (existing === undefined && baseModel === undefined) return undefined;
|
||||
if (
|
||||
existing === undefined
|
||||
&& (price(model.pricing.prompt) === undefined || price(model.pricing.completion) === undefined)
|
||||
) return undefined;
|
||||
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildBasetenModel(model, existing, baseModel),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<BasetenModel>;
|
||||
|
||||
export async function fetchBasetenModels(
|
||||
key: string,
|
||||
fetcher: typeof fetch = fetch,
|
||||
) {
|
||||
const response = await fetcher(API_ENDPOINT, {
|
||||
headers: { Authorization: `Api-Key ${key}` },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`Baseten models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return BasetenResponse.parse(await response.json());
|
||||
}
|
||||
|
||||
function price(value: string | number | undefined) {
|
||||
if (value === undefined || value === "") return undefined;
|
||||
const number = Number(value);
|
||||
return Number.isFinite(number) && number >= 0
|
||||
? Math.round(number * 1_000_000_000_000) / 1_000_000
|
||||
: undefined;
|
||||
}
|
||||
|
||||
export function buildBasetenModel(
|
||||
model: BasetenModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel = existing === undefined ? resolveBasetenBaseModel(model.id) : existing.base_model,
|
||||
): SyncedModel {
|
||||
const features = new Set(model.supported_features);
|
||||
const samplingParameters = new Set(model.supported_sampling_parameters);
|
||||
const input = modalities(model.input_modalities, existing?.modalities?.input ?? ["text"]);
|
||||
const output = modalities(model.output_modalities, existing?.modalities?.output ?? ["text"]);
|
||||
const inputCost = price(model.pricing.prompt);
|
||||
const outputCost = price(model.pricing.completion);
|
||||
const cost = inputCost !== undefined && outputCost !== undefined
|
||||
? {
|
||||
input: inputCost,
|
||||
output: outputCost,
|
||||
reasoning: existing?.cost?.reasoning,
|
||||
cache_read: existing?.cost?.cache_read,
|
||||
cache_write: existing?.cost?.cache_write,
|
||||
tiers: existing?.cost?.tiers,
|
||||
}
|
||||
: existing?.cost;
|
||||
const limit = {
|
||||
context: model.context_length,
|
||||
input: existing?.limit?.input,
|
||||
output: model.max_completion_tokens,
|
||||
};
|
||||
const values: Partial<SyncedFullModel> = {
|
||||
name: model.name ?? existing?.name,
|
||||
family: existing?.family,
|
||||
release_date: existing?.release_date,
|
||||
last_updated: existing?.last_updated,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning: features.has("reasoning") || existing?.reasoning,
|
||||
reasoning_options: existing?.reasoning_options,
|
||||
temperature: samplingParameters.has("temperature"),
|
||||
tool_call: features.has("tools") || existing?.tool_call,
|
||||
structured_output: features.has("structured_outputs") || existing?.structured_output,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: existing?.open_weights,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
};
|
||||
|
||||
if (baseModel !== undefined) {
|
||||
if (limit.context === undefined || limit.output === undefined) {
|
||||
throw new Error(`Baseten model ${model.id} has incomplete token limits required for sync`);
|
||||
}
|
||||
return factorBaseModel(baseModel, values, limit, existing?.base_model_omit);
|
||||
}
|
||||
|
||||
const required = z.object({
|
||||
name: z.string(),
|
||||
release_date: z.string(),
|
||||
last_updated: z.string(),
|
||||
open_weights: z.boolean(),
|
||||
cost: z.object({ input: z.number(), output: z.number() }),
|
||||
}).safeParse(values);
|
||||
if (!required.success) {
|
||||
throw new Error(`Baseten model ${model.id} has incomplete local metadata required for sync`);
|
||||
}
|
||||
return values as SyncedFullModel;
|
||||
}
|
||||
|
||||
export function resolveBasetenBaseModel(id: string) {
|
||||
const [prefix, ...parts] = id.split("/");
|
||||
if (prefix === undefined || parts.length === 0) return undefined;
|
||||
const canonicalPrefix = {
|
||||
"deepseek-ai": "deepseek",
|
||||
MiniMaxAI: "minimax",
|
||||
moonshotai: "moonshotai",
|
||||
nvidia: "nvidia",
|
||||
"zai-org": "zai",
|
||||
}[prefix];
|
||||
if (canonicalPrefix === undefined) return resolveCanonicalBaseModel(id);
|
||||
return resolveCanonicalBaseModel(`${canonicalPrefix}/${parts.join("/").toLowerCase()}`);
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[], fallback: Modality[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => value.toLowerCase())
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
@@ -1,219 +0,0 @@
|
||||
import { existsSync, readdirSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import { z } from "zod";
|
||||
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://llm.chutes.ai/v1/models";
|
||||
const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models");
|
||||
|
||||
const CHUTES_ORG_TO_MODEL_PROVIDER: Record<string, string | undefined> = {
|
||||
MiniMaxAI: "minimax",
|
||||
Qwen: "alibaba",
|
||||
XiaomiMiMo: "xiaomi",
|
||||
"deepseek-ai": "deepseek",
|
||||
google: "google",
|
||||
moonshotai: "moonshotai",
|
||||
openai: "openai",
|
||||
"zai-org": "zhipuai",
|
||||
};
|
||||
|
||||
const BASE_MODEL_ALIASES: Record<string, string | undefined> = {
|
||||
"google/gemma-4-31B-turbo-TEE": "google/gemma-4-31b-it",
|
||||
// "unsloth" re-hosts models from many providers, so it has no org mapping; alias the
|
||||
// ones whose canonical metadata lives under the original provider's namespace.
|
||||
"unsloth/Mistral-Nemo-Instruct-2407-TEE": "mistral/mistral-nemo",
|
||||
};
|
||||
|
||||
const Pricing = z.object({
|
||||
prompt: z.number().optional(),
|
||||
completion: z.number().optional(),
|
||||
input_cache_read: z.number().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const ChutesModel = z.object({
|
||||
id: z.string(),
|
||||
created: z.number(),
|
||||
pricing: Pricing.optional(),
|
||||
context_length: z.number().optional(),
|
||||
max_output_length: z.number().optional(),
|
||||
max_model_len: z.number().optional(),
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
supported_features: z.array(z.string()).optional(),
|
||||
supported_sampling_parameters: z.array(z.string()).optional(),
|
||||
quantization: z.string().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const ChutesResponse = z.object({
|
||||
data: z.array(ChutesModel),
|
||||
}).passthrough();
|
||||
|
||||
export type ChutesModel = z.infer<typeof ChutesModel>;
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
export const chutes = {
|
||||
id: "chutes",
|
||||
name: "Chutes",
|
||||
modelsDir: "providers/chutes/models",
|
||||
preserveBaseModels: false,
|
||||
async fetchModels() {
|
||||
const response = await fetch(API_ENDPOINT);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Chutes models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return ChutesResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildChutesModel(model, context.existing(model.id)),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<ChutesModel>;
|
||||
|
||||
export function buildChutesModel(
|
||||
model: ChutesModel,
|
||||
existing: ExistingModel | undefined,
|
||||
today = new Date().toISOString().slice(0, 10),
|
||||
): SyncedModel {
|
||||
const features = new Set(model.supported_features ?? []);
|
||||
const samplingParams = new Set(model.supported_sampling_parameters ?? []);
|
||||
const input = normalizeModalities(model.input_modalities ?? ["text"]);
|
||||
const output = normalizeModalities(model.output_modalities ?? ["text"]);
|
||||
|
||||
const attachment = input.some((value) => value !== "text");
|
||||
const reasoning = features.has("reasoning");
|
||||
const toolCall = features.has("tools");
|
||||
const structuredOutput = features.has("structured_outputs");
|
||||
// Absent sampling-parameter info, assume temperature is tunable.
|
||||
const temperature = samplingParams.size > 0 ? samplingParams.has("temperature") : true;
|
||||
|
||||
const name = existing?.name ?? humanizeModelName(model.id);
|
||||
const baseModel = resolveBaseModel(model.id);
|
||||
|
||||
const apiContext = model.context_length ?? model.max_model_len ?? 0;
|
||||
const context = apiContext > 0 ? apiContext : existing?.limit?.context ?? 0;
|
||||
const apiOutput = model.max_output_length ?? 0;
|
||||
const limit = {
|
||||
context,
|
||||
input: existing?.limit?.input,
|
||||
output: apiOutput > 0 ? apiOutput : existing?.limit?.output ?? 0,
|
||||
};
|
||||
|
||||
const cost = model.pricing?.prompt !== undefined && model.pricing?.completion !== undefined
|
||||
? {
|
||||
input: model.pricing.prompt,
|
||||
output: model.pricing.completion,
|
||||
cache_read: model.pricing.input_cache_read,
|
||||
}
|
||||
: existing?.cost;
|
||||
|
||||
const values: SyncedFullModel = {
|
||||
name,
|
||||
family: baseModel == null ? (existing?.family ?? inferFamily(model.id, name)) : existing?.family,
|
||||
release_date: existing?.release_date ?? dateFromTimestamp(model.created),
|
||||
last_updated: existing?.last_updated ?? today,
|
||||
attachment,
|
||||
reasoning,
|
||||
// Chutes' /v1/models advertises `reasoning` as a capability but exposes no parameter
|
||||
// to toggle or set its effort, so there is no provider evidence for a reasoning option.
|
||||
reasoning_options: [],
|
||||
temperature,
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput ? true : undefined,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: true,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
};
|
||||
|
||||
return baseModel == null
|
||||
? values
|
||||
: factorBaseModel(baseModel, values, limit, existing?.base_model_omit);
|
||||
}
|
||||
|
||||
function resolveBaseModel(modelId: string): string | undefined {
|
||||
return baseModelCandidates(modelId).find(canonicalExists);
|
||||
}
|
||||
|
||||
// existsSync is case-insensitive on Windows/macOS; verify the real on-disk filename case
|
||||
// so the resolved base_model matches the canonical metadata exactly (and CI on Linux).
|
||||
function canonicalExists(candidate: string): boolean {
|
||||
const file = path.join(MODELS_DIR, `${candidate}.toml`);
|
||||
if (!existsSync(file)) return false;
|
||||
try {
|
||||
return readdirSync(path.dirname(file)).includes(path.basename(file));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function baseModelCandidates(modelId: string): string[] {
|
||||
const alias = BASE_MODEL_ALIASES[modelId];
|
||||
const [org, ...modelParts] = modelId.split("/");
|
||||
if (org === undefined || modelParts.length === 0 || modelParts.join("/").endsWith("-TEE") === false) {
|
||||
return alias === undefined ? [] : [alias];
|
||||
}
|
||||
|
||||
const provider = CHUTES_ORG_TO_MODEL_PROVIDER[org];
|
||||
if (provider === undefined) {
|
||||
return alias === undefined ? [] : [alias];
|
||||
}
|
||||
|
||||
const withoutTee = modelParts.join("/").slice(0, -"-TEE".length);
|
||||
const lower = withoutTee.toLowerCase();
|
||||
// Distinct checkpoints (e.g. "-Thinking-2507") keep their own metadata — deliberately
|
||||
// not collapsed onto the generic base, which would inherit the wrong capabilities.
|
||||
const normalized = [
|
||||
withoutTee,
|
||||
lower,
|
||||
lower.replace(/-turbo$/, "-it"),
|
||||
lower.replace(/-turbo$/, ""),
|
||||
];
|
||||
|
||||
return [
|
||||
...new Set([alias, ...normalized.map((candidate) => `${provider}/${candidate}`)]).values(),
|
||||
].filter((candidate): candidate is string => candidate !== undefined);
|
||||
}
|
||||
|
||||
function normalizeModalities(values: string[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => value.toLowerCase())
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
if (result.length === 0) return ["text"];
|
||||
return [...new Set(result)];
|
||||
}
|
||||
|
||||
function humanizeModelName(modelId: string): string {
|
||||
const modelPart = modelId.split("/").at(-1) ?? modelId;
|
||||
return modelPart.replace(/-/g, " ");
|
||||
}
|
||||
|
||||
function dateFromTimestamp(timestamp: number): string {
|
||||
return new Date(timestamp * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function inferFamily(id: string, name: string) {
|
||||
const kimiFamily = inferKimiFamily(id, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const target = `${id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
.find((family) => {
|
||||
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
if (family === "o") return new RegExp(`(^|[^a-z0-9])${value}(?=\\d|$|[^a-z0-9])`).test(target);
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
|
||||
});
|
||||
}
|
||||
@@ -96,7 +96,7 @@ export const cloudflareWorkersAi = {
|
||||
},
|
||||
} satisfies SyncProvider<CloudflareModel>;
|
||||
|
||||
export function buildWorkersAiModel(
|
||||
function buildWorkersAiModel(
|
||||
model: z.infer<typeof OpenRouterModel>,
|
||||
existing: ExistingModel | undefined,
|
||||
): SyncedModel {
|
||||
@@ -108,14 +108,11 @@ export function buildWorkersAiModel(
|
||||
max_completion_tokens: existing?.limit?.output ?? model.top_provider.max_completion_tokens,
|
||||
},
|
||||
};
|
||||
const synced = {
|
||||
...buildOpenRouterModel(
|
||||
source,
|
||||
existing,
|
||||
existing?.base_model ?? resolveCloudflareBaseModel(model),
|
||||
),
|
||||
reasoning_options: existing?.reasoning_options,
|
||||
};
|
||||
const synced = buildOpenRouterModel(
|
||||
source,
|
||||
existing,
|
||||
existing?.base_model ?? resolveCloudflareBaseModel(model),
|
||||
);
|
||||
if ("base_model" in synced) return synced;
|
||||
return {
|
||||
...synced,
|
||||
|
||||
@@ -1,245 +0,0 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://router.huggingface.co/v1/models";
|
||||
|
||||
// Hugging Face org prefixes mapped to the canonical metadata prefixes understood
|
||||
// by resolveCanonicalBaseModel. Anything not listed falls back to a direct lookup.
|
||||
const CANONICAL_ORG_PREFIXES: Record<string, string> = {
|
||||
CohereLabs: "cohere",
|
||||
"deepseek-ai": "deepseek",
|
||||
google: "google",
|
||||
"meta-llama": "meta-llama",
|
||||
MiniMaxAI: "minimax",
|
||||
moonshotai: "moonshotai",
|
||||
nvidia: "nvidia",
|
||||
Qwen: "qwen",
|
||||
"stepfun-ai": "stepfun",
|
||||
XiaomiMiMo: "xiaomi",
|
||||
"zai-org": "zai",
|
||||
};
|
||||
|
||||
const HuggingFaceProvider = z.object({
|
||||
provider: z.string(),
|
||||
status: z.string(),
|
||||
context_length: z.number().int().positive().optional(),
|
||||
pricing: z.object({
|
||||
input: z.number(),
|
||||
output: z.number(),
|
||||
}).passthrough().optional(),
|
||||
throughput: z.number().nonnegative().optional(),
|
||||
first_token_latency_ms: z.number().nonnegative().optional(),
|
||||
is_free: z.boolean().optional(),
|
||||
supports_tools: z.boolean().optional(),
|
||||
supports_structured_output: z.boolean().optional(),
|
||||
is_model_author: z.boolean().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const HuggingFaceModel = z.object({
|
||||
id: z.string().min(1),
|
||||
created: z.number().optional(),
|
||||
owned_by: z.string().optional(),
|
||||
architecture: z.object({
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
}).passthrough(),
|
||||
providers: z.array(HuggingFaceProvider),
|
||||
}).passthrough();
|
||||
|
||||
export const HuggingFaceResponse = z.object({
|
||||
data: z.array(HuggingFaceModel),
|
||||
}).passthrough();
|
||||
|
||||
export type HuggingFaceModel = z.infer<typeof HuggingFaceModel>;
|
||||
export type HuggingFaceProvider = z.infer<typeof HuggingFaceProvider>;
|
||||
|
||||
export const huggingface = {
|
||||
id: "huggingface",
|
||||
name: "Hugging Face",
|
||||
modelsDir: "providers/huggingface/models",
|
||||
deleteMissing: false,
|
||||
sourceID(model) {
|
||||
return model.id;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} Hugging Face Inference Providers models were not created because their IDs could not be mapped to provider-agnostic metadata, had no live provider, or had no priced provider.`,
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
missingNotice(paths) {
|
||||
if (paths.length === 0) return [];
|
||||
return [
|
||||
`${paths.length} local Hugging Face models were absent from the Inference Providers catalog and were retained for manual lifecycle review.`,
|
||||
`Retained local paths: ${paths.map((item) => `\`${item}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
async fetchModels() {
|
||||
const headers = process.env.HF_TOKEN
|
||||
? { Authorization: `Bearer ${process.env.HF_TOKEN}` }
|
||||
: undefined;
|
||||
const response = await fetch(API_ENDPOINT, { headers });
|
||||
if (!response.ok) {
|
||||
throw new Error(`Hugging Face models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return HuggingFaceResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
if (!model.providers.some((provider) => provider.status === "live")) return undefined;
|
||||
|
||||
const existing = context.existing(model.id);
|
||||
const baseModel = existing === undefined
|
||||
? resolveHuggingFaceBaseModel(model.id)
|
||||
: existing.base_model;
|
||||
if (existing === undefined && baseModel === undefined) return undefined;
|
||||
|
||||
// The router only exposes pricing per inference provider, so a new model with
|
||||
// no priced provider cannot be created with a meaningful cost.
|
||||
const aggregate = aggregateProviders(model);
|
||||
if (existing === undefined && aggregate.cost === undefined) return undefined;
|
||||
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildHuggingFaceModel(model, existing, baseModel, aggregate),
|
||||
};
|
||||
},
|
||||
sameModel() {
|
||||
// For now the sync only creates new models; existing curated TOMLs are left
|
||||
// untouched. Treating every existing model as already in sync skips updates
|
||||
// while still allowing new files to be created.
|
||||
return true;
|
||||
},
|
||||
} satisfies SyncProvider<HuggingFaceModel>;
|
||||
|
||||
interface Aggregate {
|
||||
cost: { input: number; output: number } | undefined;
|
||||
context: number | undefined;
|
||||
tools: boolean;
|
||||
structuredOutput: boolean;
|
||||
}
|
||||
|
||||
function price(value: number) {
|
||||
return Number.isFinite(value) && value >= 0
|
||||
? Math.round(value * 1_000_000) / 1_000_000
|
||||
: undefined;
|
||||
}
|
||||
|
||||
// The router aggregates several inference providers per model and sends traffic to
|
||||
// the fastest one, so this collapses them into the route a request would actually
|
||||
// take: pricing and context from the highest-throughput provider, plus capabilities
|
||||
// advertised by any provider (a caller can always pin a slower provider).
|
||||
function aggregateProviders(model: HuggingFaceModel): Aggregate {
|
||||
const providers = model.providers.filter((provider) => provider.status === "live");
|
||||
|
||||
const byThroughput = (a: HuggingFaceProvider, b: HuggingFaceProvider) =>
|
||||
(b.throughput ?? -Infinity) - (a.throughput ?? -Infinity);
|
||||
// The provider the router routes to (fastest). Take its price when it reports one;
|
||||
// otherwise fall back to the fastest provider that does, so a new model can still
|
||||
// be costed.
|
||||
const routed = [...providers].sort(byThroughput).at(0);
|
||||
const costProvider = routed?.pricing !== undefined
|
||||
? routed
|
||||
: [...providers]
|
||||
.filter((provider): provider is HuggingFaceProvider & { pricing: { input: number; output: number } } =>
|
||||
provider.pricing !== undefined)
|
||||
.sort(byThroughput)
|
||||
.at(0);
|
||||
const input = costProvider?.pricing === undefined ? undefined : price(costProvider.pricing.input);
|
||||
const output = costProvider?.pricing === undefined ? undefined : price(costProvider.pricing.output);
|
||||
|
||||
const contexts = providers
|
||||
.map((provider) => provider.context_length)
|
||||
.filter((value): value is number => value !== undefined);
|
||||
|
||||
return {
|
||||
cost: input !== undefined && output !== undefined ? { input, output } : undefined,
|
||||
context: routed?.context_length ?? (contexts.length > 0 ? Math.max(...contexts) : undefined),
|
||||
tools: providers.some((provider) => provider.supports_tools === true),
|
||||
structuredOutput: providers.some((provider) => provider.supports_structured_output === true),
|
||||
};
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[], fallback: Modality[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => value.toLowerCase())
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
|
||||
export function buildHuggingFaceModel(
|
||||
model: HuggingFaceModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel = existing === undefined ? resolveHuggingFaceBaseModel(model.id) : existing.base_model,
|
||||
aggregate: Aggregate = aggregateProviders(model),
|
||||
): SyncedModel {
|
||||
const input = modalities(model.architecture.input_modalities, existing?.modalities?.input ?? ["text"]);
|
||||
const output = modalities(model.architecture.output_modalities, existing?.modalities?.output ?? ["text"]);
|
||||
// Pricing is curated: keep what was authored and only fall back to the router
|
||||
// (fastest route) when the local model has no cost yet.
|
||||
const cost = existing?.cost ?? aggregate.cost;
|
||||
// context/output may be unset for a freshly created base_model entry, in which case
|
||||
// factorBaseModel inherits them from the canonical metadata; the standalone-model
|
||||
// path below validates their presence at runtime.
|
||||
const limit = {
|
||||
context: existing?.limit?.context ?? aggregate.context,
|
||||
input: existing?.limit?.input,
|
||||
output: existing?.limit?.output,
|
||||
} as SyncedFullModel["limit"];
|
||||
const values: Partial<SyncedFullModel> = {
|
||||
name: existing?.name,
|
||||
family: existing?.family,
|
||||
release_date: existing?.release_date,
|
||||
last_updated: existing?.last_updated,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning: existing?.reasoning,
|
||||
reasoning_options: existing?.reasoning_options,
|
||||
temperature: existing?.temperature,
|
||||
tool_call: aggregate.tools || existing?.tool_call || undefined,
|
||||
structured_output: aggregate.structuredOutput || existing?.structured_output || undefined,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: existing?.open_weights ?? true,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
};
|
||||
|
||||
if (baseModel !== undefined) {
|
||||
return factorBaseModel(baseModel, values, limit, existing?.base_model_omit);
|
||||
}
|
||||
|
||||
// Standalone (non base_model) models require concrete booleans the router does
|
||||
// not always report; default the capability flags it leaves out.
|
||||
const full = { ...values, tool_call: values.tool_call ?? false };
|
||||
const required = z.object({
|
||||
name: z.string(),
|
||||
release_date: z.string(),
|
||||
last_updated: z.string(),
|
||||
reasoning: z.boolean(),
|
||||
open_weights: z.boolean(),
|
||||
cost: z.object({ input: z.number(), output: z.number() }),
|
||||
limit: z.object({ context: z.number(), output: z.number() }),
|
||||
}).safeParse(full);
|
||||
if (!required.success) {
|
||||
throw new Error(`Hugging Face model ${model.id} has incomplete local metadata required for sync`);
|
||||
}
|
||||
return full as SyncedFullModel;
|
||||
}
|
||||
|
||||
export function resolveHuggingFaceBaseModel(id: string) {
|
||||
const [prefix, ...parts] = id.split("/");
|
||||
if (prefix === undefined || parts.length === 0) return undefined;
|
||||
const canonicalPrefix = CANONICAL_ORG_PREFIXES[prefix];
|
||||
if (canonicalPrefix === undefined) return resolveCanonicalBaseModel(id);
|
||||
return resolveCanonicalBaseModel(`${canonicalPrefix}/${parts.join("/").toLowerCase()}`);
|
||||
}
|
||||
@@ -1,217 +0,0 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api.llmgateway.io/v1/models";
|
||||
|
||||
const Pricing = z.object({
|
||||
prompt: z.string().optional(),
|
||||
completion: z.string().optional(),
|
||||
internal_reasoning: z.string().optional(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
});
|
||||
|
||||
export const LLMGatewayModel = z.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
created: z.number(),
|
||||
family: z.string().optional(),
|
||||
architecture: z.object({
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
}),
|
||||
pricing: Pricing,
|
||||
context_length: z.number(),
|
||||
supported_parameters: z.array(z.string()),
|
||||
structured_outputs: z.boolean().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const LLMGatewayResponse = z.object({
|
||||
data: z.array(LLMGatewayModel),
|
||||
}).passthrough();
|
||||
|
||||
export type LLMGatewayModel = z.infer<typeof LLMGatewayModel>;
|
||||
|
||||
export const llmgateway = {
|
||||
id: "llmgateway",
|
||||
name: "LLM Gateway",
|
||||
modelsDir: "providers/llmgateway/models",
|
||||
async fetchModels() {
|
||||
const headers = process.env.LLMGATEWAY_API_KEY
|
||||
? { Authorization: `Bearer ${process.env.LLMGATEWAY_API_KEY}` }
|
||||
: undefined;
|
||||
const response = await fetch(API_ENDPOINT, { headers });
|
||||
if (!response.ok) {
|
||||
throw new Error(`LLM Gateway request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return LLMGatewayResponse.parse(raw).data.filter((model) => {
|
||||
const output = model.architecture.output_modalities;
|
||||
return output.length === 1 && output[0] === "text";
|
||||
});
|
||||
},
|
||||
translateModel(model, context) {
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildLLMGatewayModel(model, context.existing(model.id)),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<LLMGatewayModel>;
|
||||
|
||||
function dateFromTimestamp(timestamp: number) {
|
||||
return new Date(timestamp * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function price(value: string | undefined) {
|
||||
if (value === undefined) return undefined;
|
||||
const number = Number(value);
|
||||
return Number.isFinite(number) && number >= 0
|
||||
? Math.round(number * 1_000_000_000_000) / 1_000_000
|
||||
: undefined;
|
||||
}
|
||||
|
||||
// Cache/reasoning prices are reported as "0" when the gateway has no data; treat
|
||||
// those as unknown so we never downgrade a hand-authored value to zero.
|
||||
function nonZeroPrice(value: string | undefined) {
|
||||
const result = price(value);
|
||||
return result !== undefined && result > 0 ? result : undefined;
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[], fallback: Modality[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => value.toLowerCase())
|
||||
.map((value) => (value === "file" ? "pdf" : value))
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
|
||||
function inferFamily(model: LLMGatewayModel, name: string) {
|
||||
const kimiFamily = inferKimiFamily(model.id, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const target = `${model.id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
.find((family) => {
|
||||
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
if (family === "o") {
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=\\d|$|[^a-z0-9])`).test(target);
|
||||
}
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
|
||||
});
|
||||
}
|
||||
|
||||
function buildLLMGatewayModel(
|
||||
model: LLMGatewayModel,
|
||||
existing: ExistingModel | undefined,
|
||||
): SyncedModel {
|
||||
const prompt = price(model.pricing.prompt);
|
||||
const completion = price(model.pricing.completion);
|
||||
const reasoning = model.supported_parameters.includes("reasoning")
|
||||
|| model.supported_parameters.includes("include_reasoning");
|
||||
const context = model.context_length > 0
|
||||
? model.context_length
|
||||
: existing?.limit?.context ?? model.context_length;
|
||||
|
||||
// The gateway is authoritative for the volatile, gateway-specific data — cost
|
||||
// and served limits. Its supported_parameters / modalities are too noisy to
|
||||
// drive capability fields (it omits "tools" for flagship models yet lists
|
||||
// "temperature" for ones the catalog deliberately marks temperature=false),
|
||||
// so those stay curated: preserved from the existing entry (which, for a
|
||||
// factored model, inherits its base when the field is absent).
|
||||
const cost = prompt !== undefined && completion !== undefined
|
||||
? {
|
||||
input: prompt,
|
||||
output: completion,
|
||||
reasoning: reasoning ? nonZeroPrice(model.pricing.internal_reasoning) ?? existing?.cost?.reasoning : existing?.cost?.reasoning,
|
||||
cache_read: nonZeroPrice(model.pricing.input_cache_read) ?? existing?.cost?.cache_read,
|
||||
cache_write: nonZeroPrice(model.pricing.input_cache_write) ?? existing?.cost?.cache_write,
|
||||
tiers: existing?.cost?.tiers,
|
||||
}
|
||||
: existing?.cost;
|
||||
const limit = {
|
||||
context,
|
||||
input: existing?.limit?.input,
|
||||
output: existing?.limit?.output ?? context,
|
||||
};
|
||||
|
||||
// Existing factored model: refresh cost + limit, keep every authored override
|
||||
// as-is (undefined fields keep inheriting the base model).
|
||||
if (existing?.base_model !== undefined) {
|
||||
return factorBaseModel(
|
||||
existing.base_model,
|
||||
{
|
||||
attachment: existing.attachment,
|
||||
reasoning: existing.reasoning,
|
||||
temperature: existing.temperature,
|
||||
tool_call: existing.tool_call,
|
||||
structured_output: existing.structured_output,
|
||||
status: existing.status,
|
||||
interleaved: existing.interleaved,
|
||||
knowledge: existing.knowledge,
|
||||
modalities: existing.modalities,
|
||||
limit,
|
||||
cost,
|
||||
},
|
||||
limit,
|
||||
existing.base_model_omit,
|
||||
);
|
||||
}
|
||||
|
||||
// Existing full model: refresh cost + limit, preserve curated metadata.
|
||||
if (existing !== undefined) {
|
||||
return {
|
||||
name: existing.name ?? model.name,
|
||||
family: existing.family,
|
||||
release_date: existing.release_date ?? dateFromTimestamp(model.created),
|
||||
last_updated: existing.last_updated ?? dateFromTimestamp(model.created),
|
||||
attachment: existing.attachment ?? false,
|
||||
reasoning: existing.reasoning ?? false,
|
||||
temperature: existing.temperature ?? false,
|
||||
tool_call: existing.tool_call ?? false,
|
||||
structured_output: existing.structured_output,
|
||||
knowledge: existing.knowledge,
|
||||
open_weights: existing.open_weights ?? false,
|
||||
status: existing.status,
|
||||
interleaved: existing.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: existing.modalities ?? defaultModalities(model),
|
||||
} satisfies SyncedFullModel;
|
||||
}
|
||||
|
||||
// Brand-new model: best-effort translation from the gateway. Capability and
|
||||
// modality data are unreliable here and should be hand-reviewed.
|
||||
const { input, output } = defaultModalities(model);
|
||||
return {
|
||||
name: model.name,
|
||||
family: inferFamily(model, model.name),
|
||||
release_date: dateFromTimestamp(model.created),
|
||||
last_updated: dateFromTimestamp(model.created),
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning,
|
||||
temperature: model.supported_parameters.includes("temperature"),
|
||||
tool_call: model.supported_parameters.includes("tools")
|
||||
|| model.supported_parameters.includes("tool_choice"),
|
||||
structured_output: model.structured_outputs ?? false,
|
||||
open_weights: false,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
} satisfies SyncedFullModel;
|
||||
}
|
||||
|
||||
function defaultModalities(model: LLMGatewayModel) {
|
||||
return {
|
||||
input: modalities(model.architecture.input_modalities, ["text"]),
|
||||
output: modalities(model.architecture.output_modalities, ["text"]),
|
||||
};
|
||||
}
|
||||
@@ -2,12 +2,11 @@ import { z } from "zod";
|
||||
import { readFileSync, readdirSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import { ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
|
||||
const API_ENDPOINT = "https://openrouter.ai/api/v1/models";
|
||||
const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models");
|
||||
const MODEL_NAME_BLACKLIST = ["fable-5"];
|
||||
const modelMetadataByID = new Map<string, Record<string, unknown>>();
|
||||
const modelMetadataFilesByProvider = new Map<string, Set<string>>();
|
||||
|
||||
@@ -80,10 +79,7 @@ export const openrouter = {
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return OpenRouterResponse.parse(raw).data.filter((model) => {
|
||||
const name = `${model.id} ${model.name}`.toLowerCase();
|
||||
return MODEL_NAME_BLACKLIST.every((value) => !name.includes(value));
|
||||
});
|
||||
return OpenRouterResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
return {
|
||||
@@ -117,9 +113,6 @@ function modalities(values: string[], fallback: Modality[]): Modality[] {
|
||||
}
|
||||
|
||||
function inferFamily(model: OpenRouterModel, name: string) {
|
||||
const kimiFamily = inferKimiFamily(model.id, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const target = `${model.id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
|
||||
@@ -121,7 +121,6 @@ export function buildOvhcloudModel(
|
||||
last_updated: lastUpdated,
|
||||
attachment,
|
||||
reasoning,
|
||||
reasoning_options: reasoning ? existing?.reasoning_options : undefined,
|
||||
temperature: temperature || undefined,
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput || undefined,
|
||||
|
||||
@@ -1,241 +0,0 @@
|
||||
import { readdirSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import { z } from "zod";
|
||||
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api.venice.ai/api/v1/models?type=text";
|
||||
const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models");
|
||||
|
||||
const Capabilities = z.object({
|
||||
supportsAudioInput: z.boolean().optional(),
|
||||
supportsE2EE: z.boolean().optional(),
|
||||
supportsFunctionCalling: z.boolean().optional(),
|
||||
supportsReasoning: z.boolean().optional(),
|
||||
supportsReasoningEffort: z.boolean().optional(),
|
||||
reasoningEffortOptions: z.array(z.string()).optional(),
|
||||
supportsResponseSchema: z.boolean().optional(),
|
||||
supportsVideoInput: z.boolean().optional(),
|
||||
supportsVision: z.boolean().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const PricingTier = z.object({
|
||||
usd: z.number().nonnegative(),
|
||||
}).passthrough();
|
||||
|
||||
const ExtendedPricing = z.object({
|
||||
context_token_threshold: z.number().int().nonnegative(),
|
||||
input: PricingTier,
|
||||
output: PricingTier,
|
||||
cache_input: PricingTier.optional(),
|
||||
cache_write: PricingTier.optional(),
|
||||
}).passthrough();
|
||||
|
||||
const Pricing = z.object({
|
||||
input: PricingTier,
|
||||
output: PricingTier,
|
||||
cache_input: PricingTier.optional(),
|
||||
cache_write: PricingTier.optional(),
|
||||
extended: ExtendedPricing.optional(),
|
||||
}).passthrough();
|
||||
|
||||
const ModelSpec = z.object({
|
||||
pricing: Pricing.optional(),
|
||||
availableContextTokens: z.number().int().nonnegative(),
|
||||
maxCompletionTokens: z.number().int().nonnegative().optional(),
|
||||
capabilities: Capabilities,
|
||||
name: z.string().min(1),
|
||||
modelSource: z.string().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const VeniceModel = z.object({
|
||||
created: z.number(),
|
||||
id: z.string().min(1),
|
||||
model_spec: ModelSpec,
|
||||
}).passthrough();
|
||||
|
||||
export const VeniceResponse = z.object({
|
||||
data: z.array(VeniceModel),
|
||||
}).passthrough();
|
||||
|
||||
export type VeniceModel = z.infer<typeof VeniceModel>;
|
||||
|
||||
type ReasoningEffort = "default" | "max" | "low" | "high" | "none" | "medium" | "minimal" | "xhigh";
|
||||
|
||||
interface MetadataEntry {
|
||||
id: string;
|
||||
filename: string;
|
||||
normalizedFull: string;
|
||||
normalizedFilename: string;
|
||||
}
|
||||
|
||||
let metadataEntries: MetadataEntry[] | undefined;
|
||||
|
||||
const BASE_MODEL_ALIASES: Record<string, string> = {
|
||||
"claude-opus-4-6-fast": "anthropic/claude-opus-4-6",
|
||||
"claude-opus-4-7-fast": "anthropic/claude-opus-4-7",
|
||||
"claude-opus-4-8-fast": "anthropic/claude-opus-4-8",
|
||||
};
|
||||
|
||||
export const venice = {
|
||||
id: "venice",
|
||||
name: "Venice",
|
||||
modelsDir: "providers/venice/models",
|
||||
preserveBaseModels: false,
|
||||
async fetchModels() {
|
||||
const headers = process.env.VENICE_API_KEY
|
||||
? { Authorization: `Bearer ${process.env.VENICE_API_KEY}` }
|
||||
: undefined;
|
||||
const response = await fetch(API_ENDPOINT, { headers });
|
||||
if (!response.ok) {
|
||||
throw new Error(`Venice models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return VeniceResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
if (model.model_spec.capabilities.supportsE2EE === true) return undefined;
|
||||
const id = model.id.replaceAll("/", "-");
|
||||
const existing = context.existing(id);
|
||||
const existingBase = existing?.base_model?.startsWith("venice/") === false ? existing.base_model : undefined;
|
||||
const resolvedBase = existingBase ?? resolveVeniceBaseModel(model.id, model.model_spec.name);
|
||||
return {
|
||||
id,
|
||||
model: buildVeniceModel(model, existing, resolvedBase ?? null),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<VeniceModel>;
|
||||
|
||||
export function buildVeniceModel(
|
||||
model: VeniceModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel: string | null | undefined = existing?.base_model ?? resolveVeniceBaseModel(model.id, model.model_spec.name),
|
||||
today = new Date().toISOString().slice(0, 10),
|
||||
): SyncedModel {
|
||||
const spec = model.model_spec;
|
||||
const capabilities = spec.capabilities;
|
||||
const input = [
|
||||
"text" as const,
|
||||
...(capabilities.supportsVision ? ["image" as const] : []),
|
||||
...(capabilities.supportsAudioInput ? ["audio" as const] : []),
|
||||
...(capabilities.supportsVideoInput ? ["video" as const] : []),
|
||||
...(existing?.modalities?.input.includes("pdf") ? ["pdf" as const] : []),
|
||||
];
|
||||
const limit = {
|
||||
context: spec.availableContextTokens,
|
||||
input: existing?.limit?.input,
|
||||
output: spec.maxCompletionTokens ?? Math.floor(spec.availableContextTokens / 4),
|
||||
};
|
||||
const reasoningEfforts = capabilities.reasoningEffortOptions?.filter(isReasoningEffort);
|
||||
const reasoningOptions = reasoningEfforts?.length
|
||||
? [{ type: "effort" as const, values: reasoningEfforts }]
|
||||
: [];
|
||||
const cost = spec.pricing === undefined
|
||||
? existing?.cost
|
||||
: {
|
||||
input: spec.pricing.input.usd,
|
||||
output: spec.pricing.output.usd,
|
||||
reasoning: existing?.cost?.reasoning,
|
||||
cache_read: spec.pricing.cache_input?.usd,
|
||||
cache_write: spec.pricing.cache_write?.usd,
|
||||
input_audio: existing?.cost?.input_audio,
|
||||
output_audio: existing?.cost?.output_audio,
|
||||
tiers: spec.pricing.extended === undefined
|
||||
? existing?.cost?.tiers
|
||||
: [{
|
||||
tier: { type: "context" as const, size: spec.pricing.extended.context_token_threshold },
|
||||
input: spec.pricing.extended.input.usd,
|
||||
output: spec.pricing.extended.output.usd,
|
||||
cache_read: spec.pricing.extended.cache_input?.usd,
|
||||
cache_write: spec.pricing.extended.cache_write?.usd,
|
||||
}],
|
||||
};
|
||||
const authoritative = {
|
||||
name: spec.name,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning: capabilities.supportsReasoning === true,
|
||||
reasoning_options: reasoningOptions,
|
||||
tool_call: capabilities.supportsFunctionCalling === true,
|
||||
structured_output: capabilities.supportsResponseSchema === true ? true : undefined,
|
||||
temperature: undefined,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input: [...new Set(input)], output: ["text" as const] },
|
||||
};
|
||||
const releaseDate = new Date(model.created * 1000).toISOString().slice(0, 10);
|
||||
const values: SyncedFullModel = {
|
||||
...authoritative,
|
||||
family: baseModel == null ? inferFamily(model.id, spec.name) ?? existing?.family : existing?.family,
|
||||
release_date: releaseDate,
|
||||
last_updated: existing?.last_updated ?? today,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: spec.modelSource?.toLowerCase().includes("huggingface")
|
||||
?? existing?.open_weights
|
||||
?? false,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
};
|
||||
|
||||
return baseModel == null
|
||||
? values
|
||||
: factorBaseModel(baseModel, values, limit, existing?.base_model_omit);
|
||||
}
|
||||
|
||||
export function resolveVeniceBaseModel(id: string, name: string) {
|
||||
const alias = BASE_MODEL_ALIASES[id];
|
||||
if (alias !== undefined) return alias;
|
||||
const entries = getMetadataEntries();
|
||||
const normalizedID = normalize(id);
|
||||
const normalizedName = normalize(name);
|
||||
const ranked = [
|
||||
entries.filter((entry) => entry.normalizedFull === normalizedID),
|
||||
entries.filter((entry) => entry.normalizedFilename === normalizedID),
|
||||
entries.filter((entry) => entry.normalizedFilename === normalizedName),
|
||||
];
|
||||
return ranked.find((matches) => matches.length === 1)?.[0]?.id;
|
||||
}
|
||||
|
||||
function getMetadataEntries() {
|
||||
if (metadataEntries !== undefined) return metadataEntries;
|
||||
metadataEntries = [];
|
||||
for (const provider of readdirSync(MODELS_DIR, { withFileTypes: true })) {
|
||||
if (!provider.isDirectory()) continue;
|
||||
for (const file of readdirSync(path.join(MODELS_DIR, provider.name), { withFileTypes: true })) {
|
||||
if (!file.isFile() || !file.name.endsWith(".toml")) continue;
|
||||
const filename = file.name.slice(0, -5);
|
||||
metadataEntries.push({
|
||||
id: `${provider.name}/${filename}`,
|
||||
filename,
|
||||
normalizedFull: normalize(`${provider.name}/${filename}`),
|
||||
normalizedFilename: normalize(filename),
|
||||
});
|
||||
}
|
||||
}
|
||||
return metadataEntries;
|
||||
}
|
||||
|
||||
function normalize(value: string) {
|
||||
return value.toLowerCase().replaceAll(/[^a-z0-9]/g, "");
|
||||
}
|
||||
|
||||
function isReasoningEffort(value: string): value is ReasoningEffort {
|
||||
return ["default", "max", "low", "high", "none", "medium", "minimal", "xhigh"].includes(value);
|
||||
}
|
||||
|
||||
function inferFamily(id: string, name: string) {
|
||||
const kimiFamily = inferKimiFamily(id, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const target = `${id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
.find((family) => {
|
||||
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
if (family === "o") return new RegExp(`(^|[^a-z0-9])${value}(?=\\d|$|[^a-z0-9])`).test(target);
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
|
||||
});
|
||||
}
|
||||
@@ -1,21 +1,12 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import { ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://ai-gateway.vercel.sh/v1/models";
|
||||
|
||||
const ModelType = z.enum([
|
||||
"language",
|
||||
"embedding",
|
||||
"image",
|
||||
"video",
|
||||
"reranking",
|
||||
"transcription",
|
||||
"speech",
|
||||
"realtime",
|
||||
]);
|
||||
const ModelType = z.enum(["language", "embedding", "image", "video", "reranking"]);
|
||||
|
||||
const PricingTier = z.object({
|
||||
cost: z.string(),
|
||||
@@ -39,8 +30,8 @@ export const VercelModel = z.object({
|
||||
name: z.string(),
|
||||
created: z.number(),
|
||||
released: z.number().optional(),
|
||||
context_window: z.number().optional().default(0),
|
||||
max_tokens: z.number().optional().default(0),
|
||||
context_window: z.number(),
|
||||
max_tokens: z.number(),
|
||||
type: ModelType,
|
||||
tags: z.array(z.string()).optional().default([]),
|
||||
pricing: Pricing.optional(),
|
||||
@@ -116,17 +107,9 @@ export function buildVercelModel(model: VercelModel, existing: ExistingModel | u
|
||||
cost,
|
||||
limit: { context, input, output },
|
||||
modalities: {
|
||||
input: model.type === "transcription"
|
||||
? ["audio"]
|
||||
: model.type === "realtime"
|
||||
? ["text", "audio"]
|
||||
: ["text", tags.has("vision") ? "image" : undefined, tags.has("file-input") ? "pdf" : undefined]
|
||||
.filter((value): value is "text" | "image" | "pdf" => value !== undefined),
|
||||
output: model.type === "speech"
|
||||
? ["audio"]
|
||||
: model.type === "realtime"
|
||||
? ["text", "audio"]
|
||||
: model.type === "image"
|
||||
input: ["text", tags.has("vision") ? "image" : undefined, tags.has("file-input") ? "pdf" : undefined]
|
||||
.filter((value): value is "text" | "image" | "pdf" => value !== undefined),
|
||||
output: model.type === "image"
|
||||
? ["image"]
|
||||
: model.type === "video"
|
||||
? ["video"]
|
||||
@@ -170,9 +153,6 @@ function buildCost(pricing: VercelModel["pricing"], existing?: ExistingModel["co
|
||||
}
|
||||
|
||||
function inferFamily(modelID: string, name: string) {
|
||||
const kimiFamily = inferKimiFamily(modelID, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const targets = [modelID, name].map((value) => value.toLowerCase());
|
||||
const families = [...ModelFamilyValues].sort((a, b) => b.length - a.length);
|
||||
return families.find((family) => targets.some((target) => target.includes(family.toLowerCase())))
|
||||
@@ -196,7 +176,6 @@ function sameVercelModel(current: ExistingModel, desired: SyncedModel) {
|
||||
[current.family, desiredModel.family],
|
||||
[current.attachment, desiredModel.attachment],
|
||||
[current.reasoning, desiredModel.reasoning],
|
||||
[current.reasoning_options, desiredModel.reasoning_options],
|
||||
[current.tool_call, desiredModel.tool_call],
|
||||
[current.structured_output, desiredModel.structured_output],
|
||||
[current.open_weights, desiredModel.open_weights],
|
||||
|
||||
@@ -29,7 +29,7 @@ const XAIAPIKey = z.object({
|
||||
acls: z.array(z.string()),
|
||||
}).passthrough();
|
||||
|
||||
export type XAIModel = z.infer<typeof XAIModel>;
|
||||
type XAIModel = z.infer<typeof XAIModel>;
|
||||
|
||||
export const xai = {
|
||||
id: "xai",
|
||||
@@ -87,7 +87,7 @@ export const xai = {
|
||||
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildXAIModel(model, existing),
|
||||
model: buildModel(model, existing),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<XAIModel>;
|
||||
@@ -159,7 +159,7 @@ function cost(model: XAIModel, existing: ExistingModel) {
|
||||
};
|
||||
}
|
||||
|
||||
export function buildXAIModel(model: XAIModel, existing: ExistingModel): SyncedModel {
|
||||
function buildModel(model: XAIModel, existing: ExistingModel): SyncedModel {
|
||||
const name = existing.name;
|
||||
const attachment = existing.attachment;
|
||||
const reasoning = existing.reasoning;
|
||||
@@ -195,7 +195,6 @@ export function buildXAIModel(model: XAIModel, existing: ExistingModel): SyncedM
|
||||
last_updated: model.canonical_id === undefined ? created : lastUpdated!,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning,
|
||||
reasoning_options: existing.reasoning_options,
|
||||
temperature: existing.temperature,
|
||||
tool_call: toolCall,
|
||||
structured_output: existing.structured_output,
|
||||
|
||||
@@ -1,15 +0,0 @@
|
||||
import { expect, test } from "bun:test";
|
||||
|
||||
import { inferKimiFamily } from "../src/family.js";
|
||||
|
||||
test("Kimi family inference ignores K2 versions", () => {
|
||||
expect(inferKimiFamily("moonshotai/kimi-k2.5")).toBe("kimi-k2");
|
||||
expect(inferKimiFamily("moonshotai/kimi-k2.7-code")).toBe("kimi-k2");
|
||||
expect(inferKimiFamily("Kimi K2.6")).toBe("kimi-k2");
|
||||
});
|
||||
|
||||
test("Kimi family inference preserves thinking variants", () => {
|
||||
expect(inferKimiFamily("moonshotai/kimi-k2-thinking")).toBe("kimi-thinking");
|
||||
expect(inferKimiFamily("Kimi K2.5 Thinking")).toBe("kimi-thinking");
|
||||
expect(inferKimiFamily("moonshotai/kimi-k2.6:thinking")).toBe("kimi-thinking");
|
||||
});
|
||||
@@ -0,0 +1,35 @@
|
||||
import { expect, test } from "bun:test";
|
||||
|
||||
import { buildGoogleModel } from "../src/sync/providers/google.js";
|
||||
|
||||
test("Google sync keeps base models compact", () => {
|
||||
const synced = buildGoogleModel({
|
||||
name: "models/gemini-3-pro-image-preview",
|
||||
displayName: "Nano Banana Pro",
|
||||
inputTokenLimit: 131_072,
|
||||
outputTokenLimit: 32_768,
|
||||
temperature: 1,
|
||||
thinking: true,
|
||||
}, {
|
||||
base_model: "google/gemini-3-pro-image-preview",
|
||||
name: "Nano Banana Pro",
|
||||
family: "gemini-pro",
|
||||
release_date: "2025-11-20",
|
||||
last_updated: "2025-11-20",
|
||||
attachment: true,
|
||||
reasoning: true,
|
||||
temperature: true,
|
||||
tool_call: false,
|
||||
knowledge: "2025-01",
|
||||
open_weights: false,
|
||||
cost: { input: 2, output: 120 },
|
||||
limit: { context: 65_536, output: 32_768 },
|
||||
modalities: { input: ["text", "image"], output: ["text", "image"] },
|
||||
});
|
||||
|
||||
expect(synced).toEqual({
|
||||
base_model: "google/gemini-3-pro-image-preview",
|
||||
cost: { input: 2, output: 120 },
|
||||
limit: { context: 131_072 },
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,121 @@
|
||||
import { expect, test } from "bun:test";
|
||||
|
||||
import { preserveBaseModel } from "../src/sync/index.js";
|
||||
import { resolveCloudflareBaseModel } from "../src/sync/providers/cloudflare-workers-ai.js";
|
||||
import { buildOpenRouterModel, type OpenRouterModel } from "../src/sync/providers/openrouter.js";
|
||||
|
||||
test("OpenRouter z-ai models inherit from zhipuai metadata", () => {
|
||||
const model: OpenRouterModel = {
|
||||
id: "z-ai/glm-5.1",
|
||||
name: "Z.AI: GLM-5.1",
|
||||
created: 1_777_680_000,
|
||||
hugging_face_id: "zai-org/GLM-5.1",
|
||||
knowledge_cutoff: null,
|
||||
context_length: 200_000,
|
||||
architecture: {
|
||||
input_modalities: ["text"],
|
||||
output_modalities: ["text"],
|
||||
},
|
||||
pricing: {
|
||||
prompt: "0.0000014",
|
||||
completion: "0.0000044",
|
||||
},
|
||||
top_provider: {
|
||||
context_length: 200_000,
|
||||
max_completion_tokens: 131_072,
|
||||
},
|
||||
supported_parameters: ["tools", "tool_choice", "temperature", "structured_outputs"],
|
||||
};
|
||||
|
||||
const synced = buildOpenRouterModel(model, undefined);
|
||||
|
||||
expect("base_model" in synced ? synced.base_model : undefined).toBe("zhipuai/glm-5.1");
|
||||
});
|
||||
|
||||
test("OpenRouter-derived syncs preserve existing base model links", () => {
|
||||
const model: OpenRouterModel = {
|
||||
id: "@cf/nvidia/nemotron-3-120b-a12b",
|
||||
name: "Nemotron 3 Super 120B",
|
||||
created: 1_773_187_200,
|
||||
hugging_face_id: "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
|
||||
knowledge_cutoff: null,
|
||||
context_length: 256_000,
|
||||
architecture: {
|
||||
input_modalities: ["text"],
|
||||
output_modalities: ["text"],
|
||||
},
|
||||
pricing: {
|
||||
prompt: "0.0000005",
|
||||
completion: "0.0000015",
|
||||
},
|
||||
top_provider: {
|
||||
context_length: 256_000,
|
||||
max_completion_tokens: 256_000,
|
||||
},
|
||||
supported_parameters: ["reasoning", "tools", "temperature", "structured_outputs"],
|
||||
};
|
||||
|
||||
const synced = preserveBaseModel(buildOpenRouterModel(model, undefined), {
|
||||
base_model: "nvidia/nemotron-3-super-120b-a12b",
|
||||
base_model_omit: ["limit.input"],
|
||||
});
|
||||
|
||||
expect("base_model" in synced ? synced.base_model : undefined)
|
||||
.toBe("nvidia/nemotron-3-super-120b-a12b");
|
||||
expect("base_model_omit" in synced ? synced.base_model_omit : undefined)
|
||||
.toEqual(["limit.input"]);
|
||||
});
|
||||
|
||||
test("newly detected base models do not replace existing links", () => {
|
||||
const model: OpenRouterModel = {
|
||||
id: "z-ai/glm-5.1",
|
||||
name: "Z.AI: GLM-5.1",
|
||||
created: 1_777_680_000,
|
||||
hugging_face_id: null,
|
||||
knowledge_cutoff: null,
|
||||
context_length: 200_000,
|
||||
architecture: { input_modalities: ["text"], output_modalities: ["text"] },
|
||||
pricing: { prompt: "0.0000014", completion: "0.0000044" },
|
||||
top_provider: { context_length: 200_000, max_completion_tokens: 131_072 },
|
||||
supported_parameters: ["tools"],
|
||||
};
|
||||
const synced = buildOpenRouterModel(model, {
|
||||
base_model: "zhipuai/glm-5",
|
||||
});
|
||||
|
||||
expect("base_model" in synced ? synced.base_model : undefined).toBe("zhipuai/glm-5");
|
||||
});
|
||||
|
||||
test("undefined translated links preserve existing base model fields", () => {
|
||||
const synced = preserveBaseModel({
|
||||
base_model: undefined,
|
||||
} as never, {
|
||||
base_model: "nvidia/nemotron-3-super-120b-a12b",
|
||||
base_model_omit: ["limit.input"],
|
||||
});
|
||||
|
||||
expect("base_model" in synced ? synced.base_model : undefined)
|
||||
.toBe("nvidia/nemotron-3-super-120b-a12b");
|
||||
expect("base_model_omit" in synced ? synced.base_model_omit : undefined)
|
||||
.toEqual(["limit.input"]);
|
||||
});
|
||||
|
||||
test("new Cloudflare models discover a unique metadata base model", () => {
|
||||
const model: OpenRouterModel = {
|
||||
id: "@cf/nvidia/nemotron-3-120b-a12b",
|
||||
name: "Nemotron 3 Super 120B",
|
||||
created: 1_773_187_200,
|
||||
hugging_face_id: null,
|
||||
knowledge_cutoff: null,
|
||||
context_length: 256_000,
|
||||
architecture: { input_modalities: ["text"], output_modalities: ["text"] },
|
||||
pricing: { prompt: "0.0000005", completion: "0.0000015" },
|
||||
top_provider: { context_length: 256_000, max_completion_tokens: 256_000 },
|
||||
supported_parameters: ["reasoning"],
|
||||
};
|
||||
|
||||
expect(resolveCloudflareBaseModel(model)).toBe("nvidia/nemotron-3-super-120b-a12b");
|
||||
const synced = buildOpenRouterModel(model, undefined, resolveCloudflareBaseModel(model));
|
||||
expect("base_model" in synced ? synced.base_model : undefined)
|
||||
.toBe("nvidia/nemotron-3-super-120b-a12b");
|
||||
});
|
||||
@@ -0,0 +1,99 @@
|
||||
import { expect, test } from "bun:test";
|
||||
import path from "node:path";
|
||||
import { mkdtemp, mkdir, readlink, symlink } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
|
||||
import { syncProvider, type SyncProvider, type SyncedFullModel } from "../src/sync/index.js";
|
||||
|
||||
const model: SyncedFullModel = {
|
||||
name: "Test model",
|
||||
release_date: "2026-01-01",
|
||||
last_updated: "2026-01-01",
|
||||
attachment: false,
|
||||
reasoning: false,
|
||||
tool_call: false,
|
||||
open_weights: false,
|
||||
cost: { input: 1, output: 2 },
|
||||
limit: { context: 1_000, output: 100 },
|
||||
modalities: { input: ["text"], output: ["text"] },
|
||||
};
|
||||
|
||||
async function fixture() {
|
||||
const root = await mkdtemp(path.join(os.tmpdir(), "models-dev-sync-"));
|
||||
const modelsDir = path.join(root, "providers", "test", "models");
|
||||
await mkdir(modelsDir, { recursive: true });
|
||||
return { root, modelsDir };
|
||||
}
|
||||
|
||||
function provider(
|
||||
modelsDir: string,
|
||||
ids: string[],
|
||||
deleteMissing = true,
|
||||
preserveSymlinks = false,
|
||||
): SyncProvider<string> {
|
||||
return {
|
||||
id: "test",
|
||||
name: "Test",
|
||||
modelsDir,
|
||||
deleteMissing,
|
||||
preserveSymlinks,
|
||||
missingNotice: (paths) => paths.map((item) => `missing: ${item}`),
|
||||
async fetchModels() {
|
||||
return ids;
|
||||
},
|
||||
parseModels(raw) {
|
||||
return raw as string[];
|
||||
},
|
||||
translateModel(id) {
|
||||
return { id, model };
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
test("sync repairs a broken symlink returned by the source", async () => {
|
||||
const { modelsDir } = await fixture();
|
||||
const filePath = path.join(modelsDir, "model.toml");
|
||||
await symlink("missing.toml", filePath);
|
||||
|
||||
const result = await syncProvider(provider(modelsDir, ["model"]));
|
||||
|
||||
expect(result.created).toBe(1);
|
||||
expect(await Bun.file(filePath).text()).toContain('name = "Test model"');
|
||||
expect(readlink(filePath)).rejects.toThrow();
|
||||
});
|
||||
|
||||
test("sync preserves valid symlink aliases when configured", async () => {
|
||||
const { root, modelsDir } = await fixture();
|
||||
const targetPath = path.join(root, "target.toml");
|
||||
const filePath = path.join(modelsDir, "model.toml");
|
||||
await Bun.write(targetPath, `name = "Alias target"\n`);
|
||||
await symlink(targetPath, filePath);
|
||||
|
||||
const result = await syncProvider(provider(modelsDir, ["model"], true, true));
|
||||
|
||||
expect(result.updated).toBe(0);
|
||||
expect(await readlink(filePath)).toBe(targetPath);
|
||||
});
|
||||
|
||||
test("sync removes a broken symlink absent from the source", async () => {
|
||||
const { modelsDir } = await fixture();
|
||||
const filePath = path.join(modelsDir, "model.toml");
|
||||
await symlink("missing.toml", filePath);
|
||||
|
||||
const result = await syncProvider(provider(modelsDir, []));
|
||||
|
||||
expect(result.deleted).toBe(1);
|
||||
expect(await Bun.file(filePath).exists()).toBe(false);
|
||||
});
|
||||
|
||||
test("non-deleting sync reports missing broken symlinks", async () => {
|
||||
const { modelsDir } = await fixture();
|
||||
const filePath = path.join(modelsDir, "model.toml");
|
||||
await symlink("missing.toml", filePath);
|
||||
|
||||
const result = await syncProvider(provider(modelsDir, [], false));
|
||||
|
||||
expect(result.deleted).toBe(0);
|
||||
expect(result.notices).toEqual(["missing: model.toml"]);
|
||||
expect(await readlink(filePath)).toBe("missing.toml");
|
||||
});
|
||||
@@ -1,49 +0,0 @@
|
||||
import { expect, test } from "bun:test";
|
||||
|
||||
import { formatToml } from "../src/sync/index.js";
|
||||
|
||||
test("formats interleaved as a root field before reasoning option tables", () => {
|
||||
const content = formatToml({
|
||||
id: "example/model",
|
||||
name: "Example Model",
|
||||
release_date: "2026-01-01",
|
||||
last_updated: "2026-01-01",
|
||||
attachment: false,
|
||||
reasoning: true,
|
||||
reasoning_options: [{ type: "toggle" }],
|
||||
tool_call: true,
|
||||
interleaved: true,
|
||||
open_weights: false,
|
||||
cost: { input: 1, output: 2 },
|
||||
limit: { context: 1_000, output: 100 },
|
||||
modalities: { input: ["text"], output: ["text"] },
|
||||
});
|
||||
|
||||
expect(Bun.TOML.parse(content)).toMatchObject({
|
||||
interleaved: true,
|
||||
reasoning_options: [{ type: "toggle" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("formats empty reasoning options outside the interleaved table", () => {
|
||||
const content = formatToml({
|
||||
id: "example/model",
|
||||
name: "Example Model",
|
||||
release_date: "2026-01-01",
|
||||
last_updated: "2026-01-01",
|
||||
attachment: false,
|
||||
reasoning: true,
|
||||
reasoning_options: [],
|
||||
tool_call: true,
|
||||
interleaved: { field: "reasoning_content" },
|
||||
open_weights: false,
|
||||
cost: { input: 1, output: 2 },
|
||||
limit: { context: 1_000, output: 100 },
|
||||
modalities: { input: ["text"], output: ["text"] },
|
||||
});
|
||||
|
||||
expect(Bun.TOML.parse(content)).toMatchObject({
|
||||
interleaved: { field: "reasoning_content" },
|
||||
reasoning_options: [],
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,121 @@
|
||||
import { expect, test } from "bun:test";
|
||||
|
||||
import {
|
||||
detectChanges,
|
||||
formatToml,
|
||||
mergeModel,
|
||||
REASONING_OVERRIDES,
|
||||
} from "../script/generate-venice.js";
|
||||
|
||||
type Effort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
|
||||
const options = (...values: Effort[]) => [{ type: "effort" as const, values }];
|
||||
|
||||
function model(id: string, capabilities: Record<string, unknown>) {
|
||||
return {
|
||||
created: 1_700_000_000,
|
||||
id,
|
||||
model_spec: {
|
||||
availableContextTokens: 128_000,
|
||||
maxCompletionTokens: 32_000,
|
||||
capabilities: { supportsReasoning: true, ...capabilities },
|
||||
name: "Test Model",
|
||||
},
|
||||
object: "model",
|
||||
owned_by: "venice.ai",
|
||||
type: "text",
|
||||
};
|
||||
}
|
||||
|
||||
test("curated options survive false and stale Venice catalog metadata", () => {
|
||||
const claude = options("low", "medium", "high", "max");
|
||||
const codex = options("low", "medium", "high", "xhigh");
|
||||
const discrepancies: string[] = [];
|
||||
|
||||
expect(mergeModel(model("claude-opus-4-7", {
|
||||
supportsReasoningEffort: false,
|
||||
}), { reasoning_options: claude }, discrepancies.push.bind(discrepancies)).reasoning_options).toEqual(claude);
|
||||
expect(mergeModel(model("openai-gpt-56-codex", {
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortOptions: ["none", "low"],
|
||||
}), { reasoning_options: codex }, discrepancies.push.bind(discrepancies)).reasoning_options).toEqual(codex);
|
||||
expect(discrepancies).toHaveLength(2);
|
||||
});
|
||||
|
||||
test("documented override beats stale curated and catalog options", () => {
|
||||
const discrepancies: string[] = [];
|
||||
const merged = mergeModel(model("openai-gpt-52", {
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortOptions: ["minimal", "low", "high"],
|
||||
}), { reasoning_options: options("low", "high") }, discrepancies.push.bind(discrepancies));
|
||||
|
||||
expect(merged.reasoning_options).toEqual(options("none", "low", "medium", "high", "xhigh"));
|
||||
expect(discrepancies).toHaveLength(2);
|
||||
});
|
||||
|
||||
test("catalog fills only an uncurated new model", () => {
|
||||
const merged = mergeModel(model("new-reasoner", {
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortOptions: ["low", "high"],
|
||||
}), null);
|
||||
|
||||
expect(merged.reasoning_options).toEqual(options("low", "high"));
|
||||
});
|
||||
|
||||
test("catalog does not fill an unresolved existing model", () => {
|
||||
const existing = { reasoning: true };
|
||||
const merged = mergeModel(model("existing-unresolved-reasoner", {
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortOptions: ["low", "high"],
|
||||
}), existing);
|
||||
|
||||
expect(merged.reasoning_options).toBeUndefined();
|
||||
expect(detectChanges(existing, merged).find((change) => change.field === "reasoning_options")).toBeUndefined();
|
||||
});
|
||||
|
||||
test("catalog false without curated evidence leaves options undefined", () => {
|
||||
const merged = mergeModel(model("unknown-fixed-reasoner", {
|
||||
supportsReasoningEffort: false,
|
||||
}), null);
|
||||
|
||||
expect(merged.reasoning_options).toBeUndefined();
|
||||
expect(formatToml(merged)).not.toContain("reasoning_options");
|
||||
});
|
||||
|
||||
test("explicit curated empty options remain stable", () => {
|
||||
const existing = { reasoning_options: [] };
|
||||
const merged = mergeModel(model("curated-fixed-reasoner", {
|
||||
supportsReasoningEffort: false,
|
||||
}), existing);
|
||||
|
||||
expect(merged.reasoning_options).toEqual([]);
|
||||
expect(detectChanges({ ...existing, reasoning: true }, merged).find((change) => change.field === "reasoning_options")).toBeUndefined();
|
||||
expect(formatToml(merged)).toContain("reasoning = true\nreasoning_options = []");
|
||||
});
|
||||
|
||||
test("formatter emits nonempty options using model TOML convention", () => {
|
||||
const merged = mergeModel(model("new-reasoner", {
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortOptions: ["none", "high"],
|
||||
}), null);
|
||||
|
||||
expect(formatToml(merged)).toContain(
|
||||
'[[reasoning_options]]\ntype = "effort"\nvalues = ["none", "high"]',
|
||||
);
|
||||
});
|
||||
|
||||
test("official correction fixtures remain exact", () => {
|
||||
const expected = {
|
||||
"claude-opus-4-6": options("low", "medium", "high", "max"),
|
||||
"openai-gpt-52": options("none", "low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-52-codex": options("low", "medium", "high", "xhigh"),
|
||||
"openai-gpt-54-pro": options("medium", "high", "xhigh"),
|
||||
"gemini-3-flash-preview": options("minimal", "low", "medium", "high"),
|
||||
"kimi-k2-5": options("low", "medium", "high"),
|
||||
"qwen3-5-35b-a3b": options("low", "medium", "high"),
|
||||
"zai-org-glm-5-1": [],
|
||||
};
|
||||
|
||||
for (const [id, reasoningOptions] of Object.entries(expected)) {
|
||||
expect(REASONING_OVERRIDES[id]).toEqual(reasoningOptions);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,113 @@
|
||||
import { expect, test } from "bun:test";
|
||||
|
||||
import { buildVercelModel, type VercelModel, vercel } from "../src/sync/providers/vercel.js";
|
||||
|
||||
const model: VercelModel = {
|
||||
id: "openai/gpt-test",
|
||||
name: "GPT Test",
|
||||
created: 1_700_000_000,
|
||||
released: 1_710_000_000,
|
||||
context_window: 128_000,
|
||||
max_tokens: 32_000,
|
||||
type: "language",
|
||||
tags: ["reasoning", "tool-use", "vision", "file-input"],
|
||||
pricing: {
|
||||
input: "0.000001",
|
||||
output: "0.000004",
|
||||
input_cache_read: "0.0000001",
|
||||
},
|
||||
};
|
||||
|
||||
test("Vercel models translate gateway metadata", () => {
|
||||
const synced = buildVercelModel(model, undefined);
|
||||
|
||||
expect(synced).toMatchObject({
|
||||
name: "GPT Test",
|
||||
release_date: "2024-03-09",
|
||||
last_updated: "2024-03-09",
|
||||
attachment: true,
|
||||
reasoning: true,
|
||||
tool_call: true,
|
||||
open_weights: false,
|
||||
cost: { input: 1, output: 4, cache_read: 0.1 },
|
||||
limit: { context: 128_000, input: 96_000, output: 32_000 },
|
||||
modalities: { input: ["text", "image", "pdf"], output: ["text"] },
|
||||
});
|
||||
});
|
||||
|
||||
test("Vercel models preserve curated metadata and missing limits", () => {
|
||||
const synced = buildVercelModel({
|
||||
...model,
|
||||
context_window: 0,
|
||||
max_tokens: 0,
|
||||
}, {
|
||||
name: "Curated name",
|
||||
release_date: "2024-01-01",
|
||||
last_updated: "2025-01-01",
|
||||
reasoning_options: [{ type: "effort", values: ["low", "high"] }],
|
||||
cost: {
|
||||
input: 2,
|
||||
output: 8,
|
||||
tiers: [{
|
||||
tier: { type: "context", size: 200_000 },
|
||||
input: 3,
|
||||
output: 12,
|
||||
}],
|
||||
},
|
||||
limit: { context: 64_000, input: 48_000, output: 16_000 },
|
||||
});
|
||||
|
||||
expect(synced.name).toBe("Curated name");
|
||||
expect(synced.last_updated).toBe("2025-01-01");
|
||||
expect(synced.reasoning_options).toEqual([{ type: "effort", values: ["low", "high"] }]);
|
||||
expect(synced.cost?.tiers).toHaveLength(1);
|
||||
expect(synced.limit).toEqual({ context: 64_000, input: 48_000, output: 16_000 });
|
||||
});
|
||||
|
||||
test("Vercel non-language models use API tool capabilities", () => {
|
||||
const synced = buildVercelModel({
|
||||
...model,
|
||||
type: "image",
|
||||
tags: [],
|
||||
}, {
|
||||
tool_call: true,
|
||||
});
|
||||
|
||||
expect(synced.tool_call).toBe(false);
|
||||
});
|
||||
|
||||
test("Vercel sync includes non-language model types", () => {
|
||||
for (const [type, output] of [
|
||||
["image", ["image"]],
|
||||
["video", ["video"]],
|
||||
["reranking", ["text"]],
|
||||
] as const) {
|
||||
const source = {
|
||||
...model,
|
||||
id: `test/${type}`,
|
||||
type,
|
||||
tags: [],
|
||||
context_window: 0,
|
||||
max_tokens: 0,
|
||||
pricing: undefined,
|
||||
};
|
||||
|
||||
expect(vercel.translateModel(source, { existing: () => undefined })).toBeDefined();
|
||||
expect(buildVercelModel(source, undefined)).toMatchObject({
|
||||
tool_call: false,
|
||||
modalities: { input: ["text"], output },
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
test("Vercel models use canonical metadata when available", () => {
|
||||
const synced = buildVercelModel({
|
||||
...model,
|
||||
id: "nvidia/nemotron-3-ultra-550b-a55b",
|
||||
name: "Nemotron 3 Ultra",
|
||||
}, undefined);
|
||||
|
||||
expect("base_model" in synced ? synced.base_model : undefined)
|
||||
.toBe("nvidia/nemotron-3-ultra-550b-a55b");
|
||||
expect("last_updated" in synced ? synced.last_updated : undefined).toBeUndefined();
|
||||
});
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-10-16"
|
||||
last_updated = "2025-10-16"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-10-16"
|
||||
last_updated = "2025-10-16"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2025-05-27"
|
||||
last_updated = "2025-05-27"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-08-05"
|
||||
last_updated = "2025-08-05"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 31999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-05-22"
|
||||
last_updated = "2025-05-22"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 31999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2025-11-25"
|
||||
last_updated = "2025-11-25"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-11-25"
|
||||
last_updated = "2025-11-25"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-11-25"
|
||||
last_updated = "2025-11-25"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2026-02-06"
|
||||
last_updated = "2026-03-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-02-06"
|
||||
last_updated = "2026-03-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1024, max = 127999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-04-17"
|
||||
last_updated = "2026-04-17"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-05-22"
|
||||
last_updated = "2025-05-22"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2025-09-30"
|
||||
last_updated = "2025-09-30"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-09-30"
|
||||
last_updated = "2025-09-30"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-09-30"
|
||||
last_updated = "2025-09-30"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2026-02-18"
|
||||
last_updated = "2026-03-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-02-18"
|
||||
last_updated = "2026-03-13"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 1024, max = 63999 }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-01-20"
|
||||
last_updated = "2025-01-20"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2025-12-01"
|
||||
last_updated = "2025-12-01"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2025-07-15"
|
||||
last_updated = "2025-07-15"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-07-29"
|
||||
last_updated = "2025-07-29"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-07-29"
|
||||
last_updated = "2025-07-29"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-08-12"
|
||||
last_updated = "2025-08-12"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-09-30"
|
||||
last_updated = "2025-09-30"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-12-08"
|
||||
last_updated = "2025-12-08"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-01-20"
|
||||
last_updated = "2026-01-20"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-12-22"
|
||||
last_updated = "2025-12-22"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-03-16"
|
||||
last_updated = "2026-03-16"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-04-10"
|
||||
last_updated = "2026-04-10"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-02-12"
|
||||
last_updated = "2026-02-12"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-04-02"
|
||||
last_updated = "2026-04-02"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "toggle" }]
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-09-30"
|
||||
last_updated = "2025-09-30"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-08-08"
|
||||
last_updated = "2025-08-08"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-10-08"
|
||||
last_updated = "2025-10-08"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["high"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -3,7 +3,6 @@ release_date = "2025-08-08"
|
||||
last_updated = "2025-08-08"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = []
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-11-14"
|
||||
last_updated = "2025-11-14"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["medium"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-11-14"
|
||||
last_updated = "2025-11-14"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-12-12"
|
||||
last_updated = "2025-12-12"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["medium"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2025-12-12"
|
||||
last_updated = "2025-12-12"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-03-19"
|
||||
last_updated = "2026-03-19"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
@@ -4,7 +4,6 @@ release_date = "2026-03-19"
|
||||
last_updated = "2026-03-19"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }]
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user