Compare commits
701 Commits
issue-3177
...
issue-3879
| Author | SHA1 | Date | |
|---|---|---|---|
| d0b3056e3c | |||
| 664671e31f | |||
| b288869bbb | |||
| bfe9e932a2 | |||
| 1766ee634b | |||
| 8a8763408f | |||
| bcb11b7701 | |||
| 8853cb4de1 | |||
| 1268c4d86d | |||
| 2957c49c50 | |||
| 459813bd25 | |||
| 161d7b235c | |||
| df835625cc | |||
| 126f5f27ce | |||
| e59d9aa532 | |||
| 475e5df1de | |||
| 14471959be | |||
| b328dfe06b | |||
| c837f4d34e | |||
| 762d7feef9 | |||
| b424381291 | |||
| e6d37bdcae | |||
| bebd608155 | |||
| 6a308dfbf7 | |||
| 6455db8f76 | |||
| ebcf1c5136 | |||
| 6f4163d814 | |||
| 05c55247fa | |||
| 83b4abd291 | |||
| 5a66940016 | |||
| 3516638e90 | |||
| 214e4198af | |||
| f4ecada627 | |||
| fe06f6b0b8 | |||
| bddb089b85 | |||
| 2605c54574 | |||
| f412635d8c | |||
| b6a79f21e5 | |||
| 814f7e04e0 | |||
| 3e74f55316 | |||
| 185a4f4cc1 | |||
| b209b33ce1 | |||
| 3e72fc6374 | |||
| adfe923c2c | |||
| 7343d8b35c | |||
| 69a5617db0 | |||
| 42f0d9ff3f | |||
| 52d5045ee7 | |||
| 4faf76317a | |||
| efb5d8ea0d | |||
| 0273194a65 | |||
| a034112075 | |||
| b91080aa0e | |||
| 6fda2e07c2 | |||
| 24b7a2aa4c | |||
| d1d08289ac | |||
| 9875219078 | |||
| ec23529c0c | |||
| 62ef55a446 | |||
| 6eaf975918 | |||
| f8ac4b4fb1 | |||
| 03e2178662 | |||
| 03e495d946 | |||
| 1f5a03df40 | |||
| 0968fea09f | |||
| 1fb770040a | |||
| 1efc768ae5 | |||
| c3aab14477 | |||
| 236d2dd99a | |||
| ad211c8f8f | |||
| 38ccccc20d | |||
| cce20188e5 | |||
| 2acddd4818 | |||
| c67dbc2e02 | |||
| 790e5cb842 | |||
| 73160c42bd | |||
| ff9bc91921 | |||
| 9c249c78cb | |||
| c40d2ae925 | |||
| 71b3ca345d | |||
| 0b0414d78e | |||
| f5edd52931 | |||
| d53651e3d9 | |||
| 2e25bad01c | |||
| fcf16dcf64 | |||
| 8a61715de2 | |||
| 5d913d45eb | |||
| b975c94c43 | |||
| daafb34595 | |||
| dac8dfdf3c | |||
| efa65bbef6 | |||
| f8ab14d0d5 | |||
| 2284981d9d | |||
| 0f697e2027 | |||
| b42b2c5a43 | |||
| ee6c6dcf5f | |||
| 13f35a9f26 | |||
| 565cdf4e15 | |||
| 4dfe1920d5 | |||
| e3ae24cdd7 | |||
| 617bba5ee3 | |||
| 32ce0b9947 | |||
| 44f2b60192 | |||
| 91b5ee80f2 | |||
| 7be7cc0d3f | |||
| 342b5572a0 | |||
| 6ad4f0a5cd | |||
| ccc8c233a0 | |||
| 8b351ba0bd | |||
| 712d41fa7c | |||
| ce4d097c49 | |||
| 0b14c410cf | |||
| 7dc6b8def4 | |||
| dd79e60e32 | |||
| 04ca479ae4 | |||
| a822bef6cd | |||
| 1b69a9c4ca | |||
| 7d63db3d45 | |||
| cdf538ad29 | |||
| 317bf46e4c | |||
| 4c4cb5c8c7 | |||
| 2e815adfbb | |||
| 7894073d7d | |||
| b5d64935a1 | |||
| d2f42e9fb6 | |||
| 4ed6341d04 | |||
| 1111c28f60 | |||
| 98657bdc55 | |||
| ebcf28f7be | |||
| 2bcedfddcb | |||
| 63f35780d3 | |||
| c83101b6b0 | |||
| d84194b62d | |||
| eed1ca26ab | |||
| 5fd1300905 | |||
| 3263d558f3 | |||
| f133b51d55 | |||
| c744edfc3c | |||
| e4f8447930 | |||
| 22e4bf2620 | |||
| 4c2589610b | |||
| 39f13cbe92 | |||
| cbecae3f83 | |||
| 573c757bd2 | |||
| 8735bc603b | |||
| 6b1c5b0814 | |||
| f5ce9c666f | |||
| d3498a124c | |||
| 06af063255 | |||
| 9e9d1e7208 | |||
| 824e1f14d1 | |||
| 31ac5f5ef1 | |||
| 5c92290660 | |||
| 39f788d5e5 | |||
| 5262d3c98f | |||
| 273ab770f5 | |||
| 9053be3218 | |||
| 2f06d99dcc | |||
| 97851e4021 | |||
| 09c5d27354 | |||
| 78acd348c7 | |||
| 28d474d5b2 | |||
| 8b50f98de3 | |||
| 759ea015b2 | |||
| 5c3c6c76ff | |||
| 6301a767ef | |||
| 0360a1d239 | |||
| 44c89b7256 | |||
| 168230d28a | |||
| 06e16ed3db | |||
| d69212b0c1 | |||
| e8f9c6f2ea | |||
| d2c944568a | |||
| 023a01a015 | |||
| cd925adab8 | |||
| 4a14b64ce3 | |||
| e8e0057b12 | |||
| 6f3ae40ade | |||
| 3e4aae9ab7 | |||
| 76c38ce9c4 | |||
| 46343b601a | |||
| 0ff5e36ef7 | |||
| 8d4fe2543e | |||
| 0e4381d2a9 | |||
| 40efa93574 | |||
| 8b8c8b3d09 | |||
| c945a5f2cc | |||
| 05536f4034 | |||
| 6d6bd2c0b8 | |||
| 4a2080e2bc | |||
| 0a71b251c4 | |||
| 5b2e20cdea | |||
| f63b5ce78d | |||
| 2346146631 | |||
| 93316c1f9a | |||
| 5aef4ad9e9 | |||
| 86eb924115 | |||
| d0ac7a447b | |||
| 7f98a9100b | |||
| e0b2ffde94 | |||
| 23053dfabb | |||
| 5423b5ac78 | |||
| e3ee48788b | |||
| b013d94872 | |||
| 5736bbd70d | |||
| dc1e4c8620 | |||
| 8f12116a06 | |||
| d86fb803b6 | |||
| e3b1a320c9 | |||
| a2bf402116 | |||
| 83040e034b | |||
| 387f25aa5b | |||
| 963dc16868 | |||
| 708341dba2 | |||
| ddc2a950f3 | |||
| 889bd835ca | |||
| 6ff3db4e81 | |||
| 675e65417f | |||
| cf46b400dd | |||
| 16c0e329e2 | |||
| a63af7da95 | |||
| 1812b10f71 | |||
| 7d7b9053ac | |||
| 7138e748ef | |||
| 7d87da885e | |||
| 7c49a655cf | |||
| f2a80d69d9 | |||
| 4c9f3ef677 | |||
| c71de45508 | |||
| c356460a71 | |||
| c61a0f9163 | |||
| 180a9a5d2a | |||
| dd4571595a | |||
| 72749afeb6 | |||
| 2bf7d0ce8d | |||
| bc7639a3bd | |||
| fff8e0169f | |||
| 97af27c2d9 | |||
| 7a344d9155 | |||
| 4b96a909c3 | |||
| 0bab3e10b7 | |||
| 91eee91ce6 | |||
| 01737a716a | |||
| 93546d8e0a | |||
| 974fa55ea5 | |||
| 60129bebc6 | |||
| d4189e71d9 | |||
| 956d3ef2d6 | |||
| fb2a2353b3 | |||
| b9a7514100 | |||
| 9f1772cc66 | |||
| 8657ff5d23 | |||
| 3e786e8339 | |||
| 2698c88c5f | |||
| 84d0f286b6 | |||
| 022185b0fc | |||
| c5962a12f7 | |||
| 1d7c74d2db | |||
| 2212a7034e | |||
| 9f07595945 | |||
| 1b61b6d90a | |||
| 1b5e70c69c | |||
| 6a431dd1dd | |||
| a64f2d2cb5 | |||
| 8ad2931241 | |||
| 2673c42681 | |||
| ca21350243 | |||
| 9adacdb03f | |||
| 3bc0c6ddf7 | |||
| 6532730857 | |||
| bb22b6c979 | |||
| 0795b333b7 | |||
| 3c5c10dd7a | |||
| 0e66972fc6 | |||
| 5b615d271d | |||
| c7e82182af | |||
| 98f8abc6d9 | |||
| 66dad824a7 | |||
| 7a54b684cc | |||
| 8a45b33246 | |||
| 39dd66c818 | |||
| 03ca9dfe5f | |||
| d395d4fa5c | |||
| ab98fd3a72 | |||
| 41aed22b78 | |||
| a92ce1c6b8 | |||
| 1e752d78d7 | |||
| c6b8283234 | |||
| 36ee4db354 | |||
| a31cc6dcbd | |||
| eb199f20a2 | |||
| ef75e88af5 | |||
| 2f0d4cb935 | |||
| a5631d2abc | |||
| 195e59863b | |||
| 48c0e324f5 | |||
| f51ef90caf | |||
| eec5e00e4f | |||
| 128ab75fbe | |||
| 4800af7a79 | |||
| fa0ab66970 | |||
| 3dbcaf3b6b | |||
| 5cb56b71ca | |||
| b7888f028a | |||
| 6193ecf774 | |||
| 3c5e3ed70b | |||
| 7eaf6972b2 | |||
| bf1a01ec85 | |||
| 112f21b145 | |||
| c27b2ce71d | |||
| 800bbc1253 | |||
| 7574a882fe | |||
| 0675cf13ff | |||
| 3b2096eb78 | |||
| a049c7036b | |||
| 35709398f3 | |||
| 24abede254 | |||
| ac5223544c | |||
| 8ed112a0d7 | |||
| b9a68fdb79 | |||
| c5a634a922 | |||
| a0a26e2aef | |||
| 286bd0fc46 | |||
| 5ce4218fdd | |||
| 6deb0f9b62 | |||
| 79bcc62824 | |||
| fe316dd6c2 | |||
| 574b85ba77 | |||
| 6dcf732d1f | |||
| 417d4fc160 | |||
| 85aef3a05e | |||
| 89183a9646 | |||
| 0dcf5c4600 | |||
| f6ac2f0457 | |||
| 1d1282ffb8 | |||
| 9e26ded0d1 | |||
| ccd55d46e8 | |||
| b25ad5c651 | |||
| d0109ff70b | |||
| 3d027d7d5f | |||
| 295a483511 | |||
| 4011599eb1 | |||
| c7bf39e4e9 | |||
| a4665d1b88 | |||
| cf27076531 | |||
| 008a6b56ba | |||
| 9d1a6525c1 | |||
| c094b61135 | |||
| 3dc53bfb45 | |||
| 7fbc892898 | |||
| cb56524aa6 | |||
| d15aa5a09d | |||
| c67a5bba05 | |||
| e5931b2156 | |||
| 818b0516bd | |||
| 52db94a899 | |||
| 7d7142b080 | |||
| a0c8c6b389 | |||
| a0bb3cbda6 | |||
| a06cd852c9 | |||
| eb442c24da | |||
| c2effa3c25 | |||
| 9cfe6ecfb7 | |||
| e482704aa5 | |||
| a582dcd4a5 | |||
| df12b8724a | |||
| b4ece56d70 | |||
| 5de41e0e5f | |||
| d667b63b7f | |||
| 976ca20ebf | |||
| 8284b61189 | |||
| 5d32b5385f | |||
| 2aaff449a6 | |||
| 4975b7fb48 | |||
| 1dd66c87b2 | |||
| c79b45079c | |||
| d7fd1e1eb9 | |||
| 0e94e0644b | |||
| 0d5594e06a | |||
| 9aec93e3e2 | |||
| bfffcac592 | |||
| 2953fed88b | |||
| dc26f9bc5f | |||
| 507a68d0a1 | |||
| f7a39e320e | |||
| c3de500b5d | |||
| d479eebb47 | |||
| c8f0d25871 | |||
| d0e10d7d44 | |||
| b91f684b9b | |||
| cee485c1c4 | |||
| 4225a7adda | |||
| 725eea1b25 | |||
| a99e8b188f | |||
| af5c74ff69 | |||
| 14b8cd5612 | |||
| 1d1ef59cd1 | |||
| 7d43c564b0 | |||
| c0cfed9371 | |||
| bed4bc4f3d | |||
| 97f64dffe9 | |||
| 7eef359d6b | |||
| b463ea1300 | |||
| 6c9e0b6af0 | |||
| 5e2483bb19 | |||
| d2bb32ceb3 | |||
| 293903e820 | |||
| 1d79f0aef8 | |||
| 1f0c79af71 | |||
| f054b1f447 | |||
| 0863d10ca0 | |||
| 4d9f510e34 | |||
| fe8b0b687d | |||
| 80e891a596 | |||
| d61dc47fe4 | |||
| 7a084fd495 | |||
| 779118c76d | |||
| 8f3b8f72f0 | |||
| dc1260c4b9 | |||
| ee422deff9 | |||
| 3b11700e11 | |||
| fb7af6b667 | |||
| b059b702b8 | |||
| d6ef5792fc | |||
| 0000659c57 | |||
| a2087ac20f | |||
| 0656114326 | |||
| b49a4959dc | |||
| a0bcde206a | |||
| cc3096f9fe | |||
| a82eee57f1 | |||
| f0981b8e0c | |||
| e205675103 | |||
| e7391fee74 | |||
| 6fa1c663a8 | |||
| 288fc15ffb | |||
| 184e62b2ff | |||
| b1554e7b25 | |||
| ada66871e8 | |||
| 01462cb929 | |||
| 3e69a442ff | |||
| b11f2273e2 | |||
| d714ac65fb | |||
| 62b2b3da43 | |||
| e6caca99c8 | |||
| ae431b6535 | |||
| 4ade647a05 | |||
| 4cc0676f65 | |||
| 623d98c04f | |||
| cbb3cc97eb | |||
| 0e706c9afa | |||
| 8b25c58cec | |||
| e5be40b88b | |||
| 13da13087e | |||
| ccb24d36ce | |||
| 01aa0cdfab | |||
| e1f1d82f41 | |||
| 8b8cd78663 | |||
| 12c9f299cd | |||
| bbaff1dbd9 | |||
| a3c968e9a4 | |||
| f2fcd1c329 | |||
| 750c85633e | |||
| f79cd8b647 | |||
| 14ea63d06c | |||
| 8ade7d8d24 | |||
| 3c3c09cfe0 | |||
| 04990b4e7d | |||
| 4027cf3610 | |||
| 5885a6e726 | |||
| 324e7f0de6 | |||
| aac99b72e1 | |||
| 524524e488 | |||
| fdddf34d92 | |||
| 867fd31dc0 | |||
| bf55e760ac | |||
| a5e419dacf | |||
| c67e12e135 | |||
| 51a36c8398 | |||
| 6a6acfb4b1 | |||
| e793b58791 | |||
| 44cb2a2b2d | |||
| cecf31aa5b | |||
| 83d23e7fbd | |||
| 50bb0a0631 | |||
| e61cfe4098 | |||
| e191ef168d | |||
| 4a665d0b3a | |||
| 64cd156a32 | |||
| 275ae965fb | |||
| d27836b6d5 | |||
| cb68b9263b | |||
| 1652c012ac | |||
| e6fae89eb8 | |||
| fe9a8fcb4a | |||
| 7bec16398e | |||
| 0b877eb3c0 | |||
| 84bd139dc5 | |||
| 9397edde73 | |||
| f87e902b3b | |||
| 668a1c4360 | |||
| b9d58009d8 | |||
| d8f677ad47 | |||
| 5bf1080088 | |||
| 5806a94836 | |||
| 66dcd837fe | |||
| f08df784b5 | |||
| b875d58989 | |||
| ec996a0c4a | |||
| 4445552c8a | |||
| cfce596e3c | |||
| 774468a7a7 | |||
| b4b18ced31 | |||
| 18f001fba0 | |||
| f1e7b994a0 | |||
| c059edc1a4 | |||
| 6e87b3d24c | |||
| 2e09cb410f | |||
| fa41065a7d | |||
| ec47f923cb | |||
| ee07b91591 | |||
| cd2f6ac132 | |||
| 8613f07b5f | |||
| 0c3e1b3203 | |||
| 6cbc76a67e | |||
| 79c868244e | |||
| 6111f36aa7 | |||
| 5e9e9ac0bb | |||
| 41cf5aaf1e | |||
| c15ccd0a77 | |||
| 9d36c852d3 | |||
| 04ae13714e | |||
| 8bfa558700 | |||
| 66acdcc034 | |||
| a909f61275 | |||
| f8b382bb01 | |||
| 98975461b7 | |||
| c322362a0f | |||
| ca61e9ea27 | |||
| ac05c4ea01 | |||
| f74e16edcd | |||
| 5a21c5d14e | |||
| 71ce8da88e | |||
| 6af0231ae6 | |||
| 01e5ea975e | |||
| a08b39d342 | |||
| 929864004e | |||
| 5dd5de679e | |||
| c3a6f74ec1 | |||
| 83ffebbca7 | |||
| 084f3e1684 | |||
| 0c9b542198 | |||
| 16860bca59 | |||
| 64fed1b223 | |||
| 50cbd2d6a8 | |||
| d647ededd7 | |||
| c106790181 | |||
| 56223fd8fc | |||
| 67c29c1b6d | |||
| d98f3e9c5b | |||
| cd33292c13 | |||
| 29341789a2 | |||
| 80980b03b2 | |||
| 4a689a763a | |||
| 2f6da2e76d | |||
| 522c7cf328 | |||
| f583ccd394 | |||
| c79e5e123a | |||
| 115070da57 | |||
| a72d46bebb | |||
| 0004993aa6 | |||
| c7bdba27d4 | |||
| d637169c29 | |||
| da30cb58de | |||
| 25672f0ba7 | |||
| b0dc8d57b1 | |||
| 0aeb457b97 | |||
| d0a1ad416d | |||
| 53150a1e70 | |||
| 60b092f08e | |||
| dbb325e7ab | |||
| 7caae4cdd1 | |||
| a62ceff48b | |||
| f928b66bdd | |||
| 7b46d372a1 | |||
| a55627183d | |||
| 4198dedd57 | |||
| 174a76c3a0 | |||
| 4c8a937d43 | |||
| 2995194f70 | |||
| b9b230fbf4 | |||
| 7baf2c0970 | |||
| f5fe2210c8 | |||
| 6e4a1e55d8 | |||
| 941d97c41d | |||
| ba975a9e6f | |||
| 14431d5d10 | |||
| 56bf6e9760 | |||
| 6fbd2950ea | |||
| 4c419b3a94 | |||
| 85b0c61825 | |||
| fa5d89ef86 | |||
| d97ea65eb2 | |||
| 0b27b2f87d | |||
| c777f40587 | |||
| 02c592bd8c | |||
| b96653aa5c | |||
| 128398d56b | |||
| c1e0d97149 | |||
| b296e9819a | |||
| 4628979d6b | |||
| 4d8931d4af | |||
| 2cd05cd265 | |||
| c3ede089ed | |||
| 69b443858a | |||
| f9e613dcb3 | |||
| eb02a4d5fb | |||
| c11840c407 | |||
| a820277894 | |||
| 996ad32904 | |||
| af4e0adee1 | |||
| d099e824ac | |||
| e487ef96d3 | |||
| 70025d73bb | |||
| 57e8b1acfa | |||
| c9a85bac60 | |||
| 65f7b05891 | |||
| 6093568196 | |||
| cf2d9c134f | |||
| c2fb688fe8 | |||
| b868e9c479 | |||
| 8eeea3e02a | |||
| 5f00eb57d1 | |||
| 628d9e3170 | |||
| f15087d8ed | |||
| c252d86a33 | |||
| 7463396925 | |||
| a84c8d86e6 | |||
| ca216d4aa0 | |||
| c74a7d5540 | |||
| b5bec57ebd | |||
| f55836ab46 | |||
| 8831574def | |||
| e03f2e3d38 | |||
| 49bcef0721 | |||
| 1b89542222 | |||
| b64eb07ba9 | |||
| 3c7a10cd70 | |||
| cb70234277 | |||
| ae015bbecb | |||
| 7422d6fa48 | |||
| 21585db20a | |||
| d6e0eabbf8 | |||
| 175bd75389 | |||
| fb8f07534e | |||
| 4b50b0d338 | |||
| 0b04d0f2af | |||
| 3b3933f8f8 | |||
| c51394cdd7 | |||
| aecda0251e | |||
| 748df5f980 | |||
| 3763be6988 | |||
| 543f159a1c | |||
| 054da7d81c | |||
| ce1b441a48 | |||
| cbb3c1e732 | |||
| d5303fb7ee | |||
| af3941c5c1 | |||
| 25d8516199 | |||
| d5d3180917 | |||
| 02b5429e9b | |||
| 48b820c9fa | |||
| 4a46d08015 | |||
| 45b286a85f | |||
| 73c58b11fb | |||
| d316d6ba16 | |||
| 2c779fc8c8 | |||
| 0dd27c0da9 | |||
| c98b11d3ee | |||
| 467363a4ae | |||
| d3c3aea1d4 | |||
| a19b2008ea | |||
| 8f449a6dc8 | |||
| b2322529ae | |||
| e239a17ef1 | |||
| 0057a210b0 | |||
| 7d52d15549 | |||
| 39163abcd4 | |||
| 8f9ad03e1c | |||
| ba9e5f5403 | |||
| 6d143784e4 | |||
| 3163eaee22 | |||
| 16b7bffb0f | |||
| cdfee03695 | |||
| f5b2ae2071 | |||
| bd985e6d97 | |||
| 353af73289 | |||
| 2ee373fdf9 | |||
| de3f51e2cd | |||
| 50eeac2f73 | |||
| fcb99992a9 |
@@ -140,7 +140,7 @@ jobs:
|
||||
Failed log excerpt:
|
||||
EOF
|
||||
cat "$LOG_FILE"
|
||||
} | opencode run --agent ci-fixer -m opencode/glm-5.2 | tee "$RESPONSE_FILE"
|
||||
} | opencode run --agent ci-fixer -m opencode/grok-4.5 | tee "$RESPONSE_FILE"
|
||||
|
||||
- name: Check changed paths
|
||||
if: steps.budget.outputs.run == 'true' && steps.budget-cache.outputs.cache-hit != 'true'
|
||||
|
||||
@@ -3,23 +3,28 @@ name: Issue Fixer
|
||||
on:
|
||||
issues:
|
||||
types: [opened]
|
||||
repository_dispatch:
|
||||
types: [missing-model]
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency: issue-fixer-${{ github.event.issue.number }}
|
||||
concurrency: issue-fixer-${{ github.event.issue.number || github.event.client_payload.issue_number }}
|
||||
|
||||
jobs:
|
||||
fix:
|
||||
if: github.repository == 'anomalyco/models.dev'
|
||||
if: >-
|
||||
github.repository == 'anomalyco/models.dev'
|
||||
&& !contains(github.event.issue.labels.*.name, 'provider:openai')
|
||||
&& !contains(github.event.issue.labels.*.name, 'provider:pioneer')
|
||||
&& github.event.client_payload.provider != 'openai'
|
||||
&& github.event.client_payload.provider != 'pioneer'
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
ISSUE_NUMBER: ${{ github.event.issue.number }}
|
||||
ISSUE_TITLE: ${{ github.event.issue.title }}
|
||||
ISSUE_BODY: ${{ github.event.issue.body }}
|
||||
ISSUE_NUMBER: ${{ github.event.issue.number || github.event.client_payload.issue_number }}
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
@@ -27,6 +32,13 @@ jobs:
|
||||
with:
|
||||
ref: dev
|
||||
|
||||
- name: Load issue
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ISSUE_FILE="$RUNNER_TEMP/issue.json"
|
||||
gh issue view "$ISSUE_NUMBER" --json number,title,body,labels > "$ISSUE_FILE"
|
||||
echo "ISSUE_FILE=$ISSUE_FILE" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Install opencode
|
||||
run: curl -fsSL https://opencode.ai/install | bash
|
||||
|
||||
@@ -38,22 +50,19 @@ jobs:
|
||||
set -euo pipefail
|
||||
EVENTS_FILE="$RUNNER_TEMP/issue-fixer-events.jsonl"
|
||||
RESPONSE_FILE="$RUNNER_TEMP/issue-fixer-response.md"
|
||||
PROMPT_FILE="$RUNNER_TEMP/issue-fixer-prompt.md"
|
||||
echo "RESPONSE_FILE=$RESPONSE_FILE" >> "$GITHUB_ENV"
|
||||
|
||||
opencode run --agent issue-fixer -m opencode/glm-5.2 --format json <<EOF | tee "$EVENTS_FILE"
|
||||
A new GitHub issue was opened in anomalyco/models.dev.
|
||||
jq -r '
|
||||
"A new GitHub issue was opened in anomalyco/models.dev.\n\n"
|
||||
+ "Issue #\(.number): \(.title)\n\n"
|
||||
+ "Body:\n" + (.body // "") + "\n\n"
|
||||
+ "Decide whether this is an actionable model catalog data fix.\n\n"
|
||||
+ "If it asks for a model to be added or for factual model/provider metadata to be corrected, make the minimal TOML changes in the repository. Do not use Bash. Do not create branches, commits, comments, or pull requests yourself.\n\n"
|
||||
+ "If it is a feature request, a request to track a new kind of information, a question, or any miscellaneous non-catalog-data request, do not edit files. Respond briefly that it needs maintainer review and no automated fix was opened."
|
||||
' "$ISSUE_FILE" > "$PROMPT_FILE"
|
||||
|
||||
Issue #$ISSUE_NUMBER: $ISSUE_TITLE
|
||||
|
||||
Body:
|
||||
$ISSUE_BODY
|
||||
|
||||
Decide whether this is an actionable model catalog data fix.
|
||||
|
||||
If it asks for a model to be added or for factual model/provider metadata to be corrected, make the minimal TOML changes in the repository. Do not use Bash. Do not create branches, commits, comments, or pull requests yourself.
|
||||
|
||||
If it is a feature request, a request to track a new kind of information, a question, or any miscellaneous non-catalog-data request, do not edit files. Respond briefly that it needs maintainer review and no automated fix was opened.
|
||||
EOF
|
||||
opencode run --agent issue-fixer -m opencode/grok-4.5 --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE"
|
||||
|
||||
if ! jq -ers 'map(select(.type == "text") | .part.text) | last | select(length > 0)' "$EVENTS_FILE" > "$RESPONSE_FILE"; then
|
||||
echo "Issue fixer did not produce a final response." >&2
|
||||
@@ -74,9 +83,10 @@ jobs:
|
||||
- name: Create pull request
|
||||
if: success()
|
||||
env:
|
||||
BRANCH: issue-${{ github.event.issue.number }}
|
||||
BRANCH: issue-${{ github.event.issue.number || github.event.client_payload.issue_number }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ISSUE_TITLE="$(jq -r .title "$ISSUE_FILE")"
|
||||
|
||||
if [ -z "$(git status --porcelain)" ]; then
|
||||
if [ -s "$RESPONSE_FILE" ]; then
|
||||
|
||||
@@ -27,4 +27,4 @@ jobs:
|
||||
env:
|
||||
OPENCODE_API_KEY: ${{ secrets.OPENCODE_API_KEY }}
|
||||
with:
|
||||
model: opencode/gpt-5.5
|
||||
model: opencode/grok-4.5
|
||||
|
||||
@@ -60,7 +60,7 @@ jobs:
|
||||
RESPONSE_FILE="$RUNNER_TEMP/pr-reviewer-response.md"
|
||||
echo "RESPONSE_FILE=$RESPONSE_FILE" >> "$GITHUB_ENV"
|
||||
|
||||
opencode run --agent pr-reviewer -m opencode/glm-5.2 --format json <<'EOF' | tee "$EVENTS_FILE"
|
||||
opencode run --agent pr-reviewer -m opencode/grok-4.5 --format json <<'EOF' | tee "$EVENTS_FILE"
|
||||
Review this pull request using the trusted reviewer instructions. Start with `.pr-review/pull-request.json`, `.pr-review/diff.patch`, `AGENTS.md`, and the contributing guidance in `README.md`. Read `sync.md`, the reasoning-options audit guide, schema code, and nearby base-revision files when relevant to the changed files. Use only the read, glob, and grep tools. Return only the final review comment in the agent's required output format. Never include progress narration or passed-check summaries.
|
||||
EOF
|
||||
|
||||
|
||||
@@ -63,6 +63,7 @@ jobs:
|
||||
- name: Sync model catalogs
|
||||
run: bun models:sync ${{ matrix.provider }}
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
BASETEN_API_KEY: ${{ secrets.BASETEN_API_KEY }}
|
||||
DEEPINFRA_API_KEY: ${{ secrets.DEEPINFRA_API_KEY }}
|
||||
@@ -73,6 +74,8 @@ jobs:
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
VENICE_API_KEY: ${{ secrets.VENICE_API_KEY }}
|
||||
LLMGATEWAY_API_KEY: ${{ secrets.LLMGATEWAY_API_KEY }}
|
||||
MERGE_GATEWAY_API_KEY: ${{ secrets.MERGE_GATEWAY_API_KEY }}
|
||||
KILO_API_KEY: ${{ secrets.KILO_API_KEY }}
|
||||
GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
GOOGLE_GENERATIVE_AI_API_KEY: ${{ secrets.GOOGLE_GENERATIVE_AI_API_KEY }}
|
||||
|
||||
@@ -38,6 +38,8 @@ For model catalog changes, enforce these review rules:
|
||||
- Treat a missing compliant logo for a new provider as a merge blocker. The SVG must use `currentColor`, have no fixed size or hardcoded color, and preferably use a square `viewBox`.
|
||||
- Treat duplicated provider-agnostic metadata as a merge blocker when a matching `models/<provider>/<model>.toml` exists; the provider entry must use `base_model` and retain only provider-specific fields and overrides.
|
||||
- Treat missing `reasoning_options` on `reasoning = true` provider models as a merge blocker. Options describe controls exposed by that inference provider, not merely by the upstream model. An empty array is correct when reasoning exists but no caller control is verified.
|
||||
- Before reporting a `reasoning_options` problem, compare the proposed model with existing entries for the same underlying model that use a comparable request surface. Determine that surface from the effective `npm`, provider API shape, and any model-level provider override—not from the model family alone. Prefer native-provider examples when the target uses the native SDK (for example, an Anthropic model through `@ai-sdk/anthropic` should be compared with the Anthropic provider). Prefer established OpenAI-compatible gateway examples when the target uses an OpenAI-compatible chat-completions surface (for example, Cloudflare AI Gateway may be usefully compared with OpenRouter). Do not compare a native Anthropic route with an OpenAI-compatible gateway as though their controls were interchangeable.
|
||||
- Use those peer entries as required review context, not as values to copy mechanically or as standalone proof. Consistent same-model, same-surface examples make a proposed option more plausible and help identify likely omissions or contradictions; target-provider documentation, endpoint metadata, adapter behavior, or reproduced requests still override peer precedent. A lack of bespoke provider documentation is not by itself an action item when the target surface and strong peer examples support the proposal and the diff contains no concrete contradictory evidence. Conversely, do not accept or reject `reasoning_options = []` mechanically: explain the specific mismatch with the target API shape or comparable providers before requesting a change.
|
||||
- Do not treat absence of a sync module as a blocker. Recommend one only when a context-rich provider API can authoritatively populate model data or delete models no longer served.
|
||||
- Data-changing PRs should cite direct provider pricing, model documentation, or API references in the PR body. Missing citations are not by themselves a merge blocker, but should be reported as a low-severity request for evidence when material factual changes otherwise cannot be reviewed. Prefer first-party sources and require each citation to state what it supports.
|
||||
- You cannot fetch citation URLs. Assess whether citations are present, direct, and mapped to claims, but never claim you opened a URL or verified its contents. A URL or PR assertion alone does not prove a disputed value.
|
||||
|
||||
@@ -119,6 +119,9 @@ items are **hard blockers**; the last two are **strongly recommended** but not b
|
||||
- Latest/undated models: `@default` (`claude-opus-4-6@default.toml`)
|
||||
|
||||
### Cost Schema
|
||||
- **All `cost` values are USD per million tokens.** Never publish EUR, CNY, or other currencies.
|
||||
If a provider API or pricing page quotes another currency, convert to USD before writing the
|
||||
TOML and note the source rate/date in a top-of-file comment.
|
||||
- `cost.context_over_200k` is a nested `Cost` object for >200K token pricing
|
||||
- Cache pricing ratios: standard models use 10%/125% (read/write), regional variants may use 30%/375%
|
||||
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
description = "Poolside builds open-weight foundation models and the systems that refine and improve them."
|
||||
@@ -0,0 +1,3 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 128 128" fill="currentColor">
|
||||
<path d="m35.959 121.526c-11.8772-5.794-21.5249-14.947-27.90834-26.4686-6.23593-11.2582-8.930092-23.9574-7.798832-36.7265.256124-2.8615 2.777032-4.9741 5.639732-4.7214 2.85734.2545 4.97334 2.7778 4.72074 5.641-.94779 10.6955 1.3128 21.3362 6.538 30.7705 4.4985 8.1229 10.9417 14.84 18.8061 19.656l24.4606-50.1633c-9.5744-3.1888-17.5492-1.8007-18.2669-1.6613-.1053.0243-.2071.0414-.3106.0621-2.3841.3992-4.6901-.9038-5.6184-3.0702-1.2811-2.3919-5.1275-8.2384-9.7828-10.5094-4.6552-2.2711-11.8298-1.5385-14.1394-1.0363-1.9474.4252-3.97402-.3009-5.20405-1.8667-1.23003-1.5659-1.4658-3.7015-.5927-5.492 15.45775-31.71872 53.84575-44.93849 85.55925-29.46724 31.7136 15.47124 44.9196 53.82984 29.4886 85.53934-.016.0323-.032.0647-.049.1006-15.485 31.6834-53.8429 44.8774-85.542 29.4134zm33.8009-57.4544-24.4588 50.1594c24.6863 9.222 52.7773-1.024 65.6229-24.3097-1.806-2.7947-4.974-6.8014-8.641-8.5902-4.7375-2.3114-11.6793-1.5543-14.0641-1.0532-.3926.0933-.7839.1383-1.1773.1422-.7048.0034-1.4199-.1363-2.1061-.4355-.7114-.3114-1.3547-.781-1.874-1.386-.2968-.3495-.5421-.7317-.7393-1.1395-.1533-.3062-3.9466-7.6667-12.5659-13.3893zm-38.7651-29.0902c3.9831 1.9431 7.2244 5.0332 9.6483 7.947 7.496-11.4666 17.6688-20.1275 25.527-25.7116 2.9201-2.0736 5.9436-4.0123 8.8552-5.6852-20.4537-4.29467-41.8903 3.8115-54.3197 20.8782 3.2899.2252 6.9209.9284 10.2892 2.5716zm67.5712-11.9611c.4747 3.3248.8105 6.8979.9729 10.4798.4384 9.6049-.1169 22.9086-4.5038 35.8475 3.6589.0981 7.9139.7451 11.8069 2.6443 3.476 1.6959 6.39 4.2614 8.684 6.8223 5.855-20.3405-.95-42.2864-16.9617-55.7903zm-28.7702 29.1142c7.1932 3.5091 12.3927 8.1776 15.9169 12.2023 5.733-18.6289 3.2338-39.4757 1.1469-47.1965-7.3675 3.1085-25.3335 13.9715-36.4767 29.961 5.3459.2981 12.2232 1.5257 19.4129 5.0332z"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 1.8 KiB |
@@ -3,7 +3,7 @@ description = "Earlier Qwen multimodal workhorse for million-token agent and doc
|
||||
family = "qwen"
|
||||
release_date = "2026-04-02"
|
||||
last_updated = "2026-04-02"
|
||||
attachment = false
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
|
||||
@@ -16,3 +16,67 @@ output = 65_536
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 80.4
|
||||
metric = "resolved"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 60.6
|
||||
metric = "resolve rate"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Multilingual"
|
||||
score = 78.3
|
||||
metric = "resolve rate"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 69.7
|
||||
metric = "success rate"
|
||||
harness = "Terminus-2"
|
||||
version = "2.0"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 92.4
|
||||
metric = "accuracy"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 41.4
|
||||
metric = "accuracy"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SciCode"
|
||||
score = 53.5
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 76.4
|
||||
metric = "success rate"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "NL2Repo"
|
||||
score = 47.2
|
||||
harness = "Claude Code"
|
||||
source = "https://qwen.ai/blog?id=qwen3.7"
|
||||
date = "2026-05-19"
|
||||
|
||||
@@ -3,7 +3,7 @@ description = "Multimodal Qwen workhorse for long-context agents, visual inputs,
|
||||
family = "qwen"
|
||||
release_date = "2026-06-02"
|
||||
last_updated = "2026-06-02"
|
||||
attachment = false
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
@@ -15,5 +15,5 @@ context = 1_000_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
# Sources (accessed 2026-07-20):
|
||||
# https://docs.qwencloud.com/token-plan/personal/token-plan-personal-overview
|
||||
# https://platform.qianwenai.com/docs/token-plan/personal/token-plan-personal-overview
|
||||
# https://docs.qwencloud.com/developer-guides/getting-started/text-generation-models
|
||||
# https://platform.qianwenai.com/docs/developer-guides/getting-started/text-generation-models
|
||||
# https://docs.qwencloud.com/developer-guides/clients-and-developer-tools/opencode
|
||||
# https://platform.qianwenai.com/docs/developer-guides/clients-and-developer-tools/opencode
|
||||
# https://docs.qwencloud.com/developer-guides/clients-and-developer-tools/kilo-cli
|
||||
# https://platform.qianwenai.com/docs/developer-guides/clients-and-developer-tools/kilo-cli
|
||||
# https://github.com/QwenLM/qwen-code/issues/7198
|
||||
# https://github.com/QwenLM/qwen-code/pull/7199
|
||||
|
||||
name = "Qwen3.8 Max Preview"
|
||||
description = "Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows"
|
||||
family = "qwen"
|
||||
release_date = "2026-07-19"
|
||||
last_updated = "2026-07-19"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 131_072
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -17,3 +17,70 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 80.3
|
||||
metric = "resolve rate"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 95
|
||||
metric = "resolved"
|
||||
source = "https://benchlm.ai/benchmarks/sweVerified"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 88.0
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 59
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 64.5
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 85
|
||||
metric = "success rate"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierCode"
|
||||
score = 29.3
|
||||
metric = "pass rate"
|
||||
variant = "high effort"
|
||||
dataset = "Diamond"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval-AA"
|
||||
score = 1932
|
||||
metric = "Elo"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "AutomationBench"
|
||||
score = 17.4
|
||||
metric = "success rate"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
@@ -142,3 +142,33 @@ harness = "Claude Code"
|
||||
variant = "medium"
|
||||
version = "2.1"
|
||||
source = "https://artificialanalysis.ai/agents/coding-agents"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 94.2
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 46.9
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 54.7
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 78.0
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
@@ -33,3 +33,41 @@ harness = "Terminus-2"
|
||||
version = "2.1"
|
||||
source = "https://www.anthropic.com/news/claude-opus-4-8"
|
||||
date = "2026-05-28"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 88.6
|
||||
metric = "resolved"
|
||||
source = "https://benchlm.ai/benchmarks/sweVerified"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 49.8
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 57.9
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 83.4
|
||||
metric = "success rate"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierCode"
|
||||
score = 13.4
|
||||
metric = "pass rate"
|
||||
variant = "high effort"
|
||||
dataset = "Diamond"
|
||||
source = "https://www.anthropic.com/news/claude-fable-5-mythos-5"
|
||||
date = "2026-06-09"
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
name = "Claude Opus 5"
|
||||
description = "Strongest Claude Opus model for coding, agents, and professional work"
|
||||
family = "claude-opus"
|
||||
release_date = "2026-07-24"
|
||||
last_updated = "2026-07-24"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
open_weights = false
|
||||
knowledge = "2026-05"
|
||||
|
||||
[limit]
|
||||
context = 1_000_000
|
||||
output = 128_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -72,3 +72,35 @@ harness = "Claude Code"
|
||||
variant = "medium"
|
||||
version = "2.1"
|
||||
source = "https://artificialanalysis.ai/agents/coding-agents"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 67.0
|
||||
metric = "success rate"
|
||||
harness = "Terminus-2"
|
||||
version = "2.1"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 34.6
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 46.8
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 78.5
|
||||
metric = "success rate"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
@@ -17,3 +17,56 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 85.2
|
||||
metric = "resolved"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 63.2
|
||||
metric = "resolve rate"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Multilingual"
|
||||
score = 78.3
|
||||
metric = "resolve rate"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 80.4
|
||||
metric = "success rate"
|
||||
harness = "Terminus-2"
|
||||
version = "2.1"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 81.2
|
||||
metric = "success rate"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 84.7
|
||||
metric = "accuracy"
|
||||
variant = "single agent"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierCode"
|
||||
score = 38.8
|
||||
metric = "pass rate"
|
||||
version = "v1"
|
||||
source = "https://www.anthropic.com/news/claude-sonnet-5"
|
||||
date = "2026-06-30"
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
# https://huggingface.co/CohereLabs/aya-expanse-32b
|
||||
name = "Aya Expanse 32B"
|
||||
description = "Open multilingual model optimized for generation across 23 languages"
|
||||
release_date = "2024-10-24"
|
||||
last_updated = "2024-10-24"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
license = "CC-BY-NC-4.0"
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
output = 4_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/aya-expanse-32b"
|
||||
@@ -0,0 +1,23 @@
|
||||
# https://huggingface.co/CohereLabs/aya-expanse-8b
|
||||
name = "Aya Expanse 8B"
|
||||
description = "Compact open multilingual model optimized for generation across 23 languages"
|
||||
release_date = "2024-10-24"
|
||||
last_updated = "2024-10-24"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
license = "CC-BY-NC-4.0"
|
||||
|
||||
[limit]
|
||||
context = 8_000
|
||||
output = 4_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/aya-expanse-8b"
|
||||
@@ -0,0 +1,23 @@
|
||||
# https://huggingface.co/CohereLabs/aya-vision-32b
|
||||
name = "Aya Vision 32B"
|
||||
description = "Open multilingual vision model for OCR, visual reasoning, and image question answering"
|
||||
release_date = "2025-03-04"
|
||||
last_updated = "2025-05-14"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
license = "CC-BY-NC-4.0"
|
||||
|
||||
[limit]
|
||||
context = 16_000
|
||||
output = 4_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/aya-vision-32b"
|
||||
@@ -0,0 +1,23 @@
|
||||
# https://huggingface.co/CohereLabs/aya-vision-8b
|
||||
name = "Aya Vision 8B"
|
||||
description = "Compact open multilingual vision model for OCR and visual question answering"
|
||||
release_date = "2025-03-04"
|
||||
last_updated = "2025-05-14"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
license = "CC-BY-NC-4.0"
|
||||
|
||||
[limit]
|
||||
context = 16_000
|
||||
output = 4_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/aya-vision-8b"
|
||||
@@ -0,0 +1,25 @@
|
||||
# https://docs.cohere.com/docs/command-a-reasoning
|
||||
# https://huggingface.co/CohereLabs/c4ai-command-a-reasoning-08-2025
|
||||
name = "Command A Reasoning"
|
||||
description = "Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows"
|
||||
family = "command-a"
|
||||
release_date = "2025-08-21"
|
||||
last_updated = "2025-08-21"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
knowledge = "2024-06-01"
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 256_000
|
||||
output = 32_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/c4ai-command-a-reasoning-08-2025"
|
||||
@@ -0,0 +1,25 @@
|
||||
# https://docs.cohere.com/docs/models
|
||||
# https://huggingface.co/CohereLabs/c4ai-command-a-translate-08-2025
|
||||
name = "Command A Translate"
|
||||
description = "Translation model for multilingual conversion, localization, and cross-language workflows"
|
||||
family = "command-a"
|
||||
release_date = "2025-08-28"
|
||||
last_updated = "2025-08-28"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-06-01"
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 8_000
|
||||
output = 8_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/c4ai-command-a-translate-08-2025"
|
||||
@@ -0,0 +1,25 @@
|
||||
# https://docs.cohere.com/docs/command-a-vision
|
||||
# https://huggingface.co/CohereLabs/c4ai-command-a-vision-07-2025
|
||||
name = "Command A Vision"
|
||||
description = "Cohere vision model for multilingual document analysis, OCR, and image understanding"
|
||||
family = "command-a"
|
||||
release_date = "2025-07-31"
|
||||
last_updated = "2025-07-31"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-06-01"
|
||||
tool_call = false
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
output = 8_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/c4ai-command-a-vision-07-2025"
|
||||
@@ -0,0 +1,25 @@
|
||||
# https://huggingface.co/CohereLabs/c4ai-command-r7b-arabic-02-2025
|
||||
# https://docs.cohere.com/changelog/command-r7b-arabic
|
||||
name = "Command R7B Arabic"
|
||||
description = "Open Command R model optimized for Arabic enterprise chat, RAG, and cultural knowledge"
|
||||
family = "command-r"
|
||||
release_date = "2025-02-27"
|
||||
last_updated = "2025-02-27"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
knowledge = "2024-06-01"
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
output = 4_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/CohereLabs/c4ai-command-r7b-arabic-02-2025"
|
||||
@@ -18,3 +18,47 @@ output = 64_000
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 67.6
|
||||
metric = "resolved"
|
||||
harness = "SWE-agent"
|
||||
source = "https://huggingface.co/CohereLabs/North-Mini-Code-1.0"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 40.2
|
||||
metric = "resolve rate"
|
||||
harness = "SWE-agent"
|
||||
source = "https://huggingface.co/CohereLabs/North-Mini-Code-1.0"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Intelligence Index"
|
||||
score = 27.6
|
||||
metric = "index score"
|
||||
source = "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Coding Index"
|
||||
score = 33.4
|
||||
metric = "index score"
|
||||
source = "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval-AA"
|
||||
score = 14
|
||||
metric = "win rate"
|
||||
source = "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model"
|
||||
date = "2026-06-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "τ²-Bench Telecom"
|
||||
score = 37
|
||||
metric = "success rate"
|
||||
source = "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model"
|
||||
date = "2026-06-09"
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/deep-research-max-preview-04-2026
|
||||
# - https://ai.google.dev/gemini-api/docs/deep-research
|
||||
# - https://blog.google/innovation-and-ai/models-and-research/gemini-models/next-generation-gemini-deep-research/
|
||||
|
||||
name = "Deep Research Max Preview"
|
||||
description = "Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports"
|
||||
family = "gemini-pro"
|
||||
release_date = "2026-04-21"
|
||||
last_updated = "2026-04-21"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text", "image"]
|
||||
@@ -0,0 +1,24 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/deep-research-preview-04-2026
|
||||
# - https://ai.google.dev/gemini-api/docs/deep-research
|
||||
# - https://blog.google/innovation-and-ai/models-and-research/gemini-models/next-generation-gemini-deep-research/
|
||||
|
||||
name = "Gemini Deep Research Preview"
|
||||
description = "Agentic model for autonomous multi-step research, synthesis, and cited reports"
|
||||
family = "gemini-pro"
|
||||
release_date = "2026-04-21"
|
||||
last_updated = "2026-04-21"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text", "image"]
|
||||
@@ -0,0 +1,27 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/gemini-2.5-computer-use-preview-10-2025
|
||||
# (model id, modalities text+image in / text out, input 128000, output 64000, latest update Oct 2025)
|
||||
# - https://ai.google.dev/gemini-api/docs/computer-use
|
||||
# (legacy computer-use model; tool/function actions; still listed as available)
|
||||
# - https://blog.google/innovation-and-ai/models-and-research/google-deepmind/gemini-computer-use-model/
|
||||
# (public preview 2025-10-07; built on Gemini 2.5 Pro visual + reasoning)
|
||||
|
||||
name = "Gemini 2.5 Computer Use Preview"
|
||||
description = "Specialized Gemini 2.5 model for browser-control agents that automate UI tasks"
|
||||
family = "gemini-pro"
|
||||
release_date = "2025-10-07"
|
||||
last_updated = "2025-10-07"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
@@ -7,7 +7,7 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = false
|
||||
knowledge = "2025-06"
|
||||
knowledge = "2024-06"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
name = "Nano Banana Pro"
|
||||
description = "Nano Banana Pro for higher-fidelity image generation and design-heavy edits"
|
||||
family = "gemini-pro"
|
||||
release_date = "2026-05-28"
|
||||
last_updated = "2026-05-28"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = false
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 65_536
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text", "image"]
|
||||
@@ -0,0 +1,19 @@
|
||||
name = "Nano Banana 2"
|
||||
description = "Image model for prompt-driven generation, editing, and visual design workflows"
|
||||
family = "gemini-flash"
|
||||
release_date = "2026-05-28"
|
||||
last_updated = "2026-05-28"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = false
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "pdf"]
|
||||
output = ["text", "image"]
|
||||
@@ -0,0 +1,25 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image
|
||||
# - https://ai.google.dev/gemini-api/docs/image-generation
|
||||
# - https://deepmind.google/models/model-cards/gemini-3-1-flash-lite-image/
|
||||
# - https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-1-flash-lite-image
|
||||
name = "Nano Banana 2 Lite"
|
||||
description = "Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing"
|
||||
family = "gemini-flash-lite"
|
||||
release_date = "2026-06-30"
|
||||
last_updated = "2026-06-30"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = false
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 65_536
|
||||
output = 4_096
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text", "image"]
|
||||
@@ -0,0 +1,24 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-live-preview
|
||||
# - https://blog.google/innovation-and-ai/technology/developers-tools/build-with-gemini-3-1-flash-live/
|
||||
# - https://deepmind.google/models/model-cards/gemini-3-1-flash-audio/
|
||||
name = "Gemini 3.1 Flash Live Preview"
|
||||
description = "High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications"
|
||||
family = "gemini-flash"
|
||||
release_date = "2026-03-26"
|
||||
last_updated = "2026-03-26"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = false
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio"]
|
||||
output = ["text", "audio"]
|
||||
@@ -0,0 +1,23 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
|
||||
# - https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-1-flash-tts/
|
||||
|
||||
name = "Gemini 3.1 Flash TTS Preview"
|
||||
description = "Low-latency speech generation with steerable prompts and expressive audio tags"
|
||||
family = "gemini-flash"
|
||||
release_date = "2026-04-15"
|
||||
last_updated = "2026-04-15"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 8_192
|
||||
output = 16_384
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["audio"]
|
||||
@@ -96,3 +96,62 @@ harness = "Gemini CLI"
|
||||
variant = "high"
|
||||
version = "2.1"
|
||||
source = "https://artificialanalysis.ai/agents/coding-agents"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 94.3
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 44.4
|
||||
metric = "accuracy"
|
||||
dataset = "full set, text + MM"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ARC-AGI-2"
|
||||
score = 77.1
|
||||
metric = "accuracy"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 80.5
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 78.2
|
||||
metric = "success rate"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 76.2
|
||||
metric = "success rate"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "CharXiv Reasoning"
|
||||
score = 83.3
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval-AA"
|
||||
score = 1314
|
||||
metric = "Elo"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
name = "Gemini 3.5 Flash Lite"
|
||||
description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost"
|
||||
family = "gemini-flash-lite"
|
||||
release_date = "2026-07-21"
|
||||
last_updated = "2026-07-21"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-03"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -18,3 +18,80 @@ output = 65_536
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 76.2
|
||||
metric = "success rate"
|
||||
harness = "Terminus-2"
|
||||
version = "2.1"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 55.1
|
||||
metric = "resolve rate"
|
||||
variant = "single attempt"
|
||||
dataset = "public"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 83.6
|
||||
metric = "success rate"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon"
|
||||
score = 56.5
|
||||
metric = "success rate"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 78.4
|
||||
metric = "success rate"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 83.6
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "CharXiv Reasoning"
|
||||
score = 84.2
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 40.2
|
||||
metric = "accuracy"
|
||||
dataset = "full set, text + MM"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ARC-AGI-2"
|
||||
score = 72.1
|
||||
metric = "accuracy"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval-AA"
|
||||
score = 1656
|
||||
metric = "Elo"
|
||||
source = "https://deepmind.google/models/gemini/flash/"
|
||||
date = "2026-05-19"
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/gemini-3.5-live-translate-preview
|
||||
# - https://ai.google.dev/gemini-api/docs/live-api/live-translate
|
||||
# - https://deepmind.google/models/model-cards/gemini-3-5-audio/
|
||||
# - https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-live-3-5-translate/
|
||||
name = "Gemini 3.5 Live Translate Preview"
|
||||
description = "Low-latency audio-to-audio model for real-time speech translation across 70+ languages"
|
||||
family = "gemini-pro"
|
||||
release_date = "2026-06-09"
|
||||
last_updated = "2026-06-09"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["audio"]
|
||||
output = ["audio", "text"]
|
||||
@@ -0,0 +1,20 @@
|
||||
name = "Gemini 3.6 Flash"
|
||||
description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost"
|
||||
family = "gemini-flash"
|
||||
release_date = "2026-07-21"
|
||||
last_updated = "2026-07-21"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2026-03"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,19 @@
|
||||
name = "Gemini Embedding 2"
|
||||
description = "Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space"
|
||||
family = "gemini"
|
||||
release_date = "2026-04-22"
|
||||
last_updated = "2026-04-22"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
knowledge = "2025-11"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 8_192
|
||||
output = 3_072
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "audio", "video", "pdf"]
|
||||
output = ["text"]
|
||||
@@ -1,8 +1,9 @@
|
||||
# Tracks the current Gemini Flash release (gemini-3.5-flash).
|
||||
name = "Gemini Flash Latest"
|
||||
description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost"
|
||||
family = "gemini-flash"
|
||||
release_date = "2025-09-25"
|
||||
last_updated = "2025-09-25"
|
||||
release_date = "2026-05-19"
|
||||
last_updated = "2026-05-19"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -16,5 +17,5 @@ context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "audio", "video", "pdf"]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
# Tracks the current Gemini Flash-Lite release (gemini-3.1-flash-lite).
|
||||
name = "Gemini Flash-Lite Latest"
|
||||
description = "Low-latency Gemini model for high-volume multimodal and agent workloads"
|
||||
family = "gemini-flash-lite"
|
||||
release_date = "2025-09-25"
|
||||
last_updated = "2025-09-25"
|
||||
release_date = "2026-05-07"
|
||||
last_updated = "2026-05-07"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
@@ -16,5 +17,5 @@ context = 1_048_576
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "audio", "video", "pdf"]
|
||||
input = ["text", "image", "video", "audio", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
@@ -15,3 +15,10 @@ output = 57_920
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["video"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "LMArena Text-to-Video Arena"
|
||||
score = 1527
|
||||
metric = "Elo"
|
||||
source = "https://venturebeat.com/technology/googles-gemini-omni-flash-hits-the-api-turning-enterprise-video-production-into-a-conversation"
|
||||
date = "2026-06-30"
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/gemini-robotics-er-1.6-preview
|
||||
# - https://ai.google.dev/gemini-api/docs/robotics-overview
|
||||
# - https://blog.google/innovation-and-ai/models-and-research/google-deepmind/gemini-robotics-er-1-6
|
||||
# - https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-Robotics-ER-1-6-Model-Card.pdf
|
||||
|
||||
name = "Gemini Robotics-ER 1.6 Preview"
|
||||
description = "Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics"
|
||||
family = "gemini"
|
||||
release_date = "2026-04-14"
|
||||
last_updated = "2026-04-14"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
knowledge = "2025-01"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video", "audio"]
|
||||
output = ["text"]
|
||||
@@ -0,0 +1,25 @@
|
||||
# https://ai.google.dev/gemini-api/docs/models/lyria-3-clip-preview
|
||||
# https://ai.google.dev/gemini-api/docs/music-generation
|
||||
# https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/lyria/lyria-3
|
||||
# https://ai.google.dev/gemini-api/docs/pricing
|
||||
# https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing
|
||||
|
||||
name = "Lyria 3 Clip Preview"
|
||||
description = "Music generation model for short 30-second clips, loops, and previews from text or image prompts"
|
||||
family = "lyria"
|
||||
release_date = "2026-03-25"
|
||||
last_updated = "2026-03-25"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
structured_output = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 65_536
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text", "audio"]
|
||||
@@ -0,0 +1,26 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/lyria-3-pro-preview — model card: text+image in; audio+lyrics text out; input token limit 131,072; no tools/thinking/structured output/caching
|
||||
# - https://ai.google.dev/gemini-api/docs/music-generation — full-length song generation; MP3 (WAV optional); lyrics/structure text in responses
|
||||
# - https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/lyria/lyria-3 — release_date 2026-03-25; preview; text+image input; audio output; max ~184s
|
||||
# - https://blog.google/innovation-and-ai/technology/developers-tools/lyria-3-developers/ — public preview announcement (2026-03-25)
|
||||
# Output token limit not published on the first-party model card; 8_192 retained from LiteLLM cost map pending Models API sync overwrite.
|
||||
|
||||
name = "Lyria 3 Pro Preview"
|
||||
description = "Music generation model for full-length songs from text or images with vocals and structure"
|
||||
family = "lyria"
|
||||
release_date = "2026-03-25"
|
||||
last_updated = "2026-03-25"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
structured_output = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 131_072
|
||||
output = 8_192
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text", "audio"]
|
||||
@@ -0,0 +1,18 @@
|
||||
name = "Veo 3.1 Fast Preview"
|
||||
description = "Video model for prompt-guided generation, editing, and motion workflows"
|
||||
family = "veo"
|
||||
release_date = "2025-10-15"
|
||||
last_updated = "2026-01-01"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_024
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["video"]
|
||||
@@ -0,0 +1,23 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/veo-3.1-generate-preview
|
||||
# - https://ai.google.dev/gemini-api/docs/veo
|
||||
# - https://developers.googleblog.com/introducing-veo-3-1-and-new-creative-capabilities-in-the-gemini-api
|
||||
|
||||
name = "Veo 3.1 Preview"
|
||||
description = "Video model for prompt-guided generation, editing, and motion workflows"
|
||||
family = "veo"
|
||||
release_date = "2025-10-15"
|
||||
last_updated = "2026-01"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_024
|
||||
output = 1
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["video"]
|
||||
@@ -0,0 +1,26 @@
|
||||
# Sources:
|
||||
# - https://ai.google.dev/gemini-api/docs/models/veo-3.1-lite-generate-preview
|
||||
# (model code, text+image input, video+audio output, 1,024 text input tokens, March 2026 update)
|
||||
# - https://blog.google/innovation-and-ai/technology/ai/veo-3-1-lite/
|
||||
# (release 2026-03-31; text-to-video and image-to-video; 720p/1080p; 4s/6s/8s)
|
||||
# - https://ai.google.dev/gemini-api/docs/pricing
|
||||
# (Veo 3.1 Lite paid-tier per-second video pricing; not token-based — cost omitted)
|
||||
|
||||
name = "Veo 3.1 Lite Preview"
|
||||
description = "Video model for prompt-guided generation, editing, and motion workflows"
|
||||
family = "veo"
|
||||
release_date = "2026-03-31"
|
||||
last_updated = "2026-03-31"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_024
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["video"]
|
||||
@@ -16,3 +16,53 @@ output = 131_072
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 59.5
|
||||
metric = "resolve rate"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Multilingual"
|
||||
score = 77.3
|
||||
metric = "resolve rate"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 70.8
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 88.9
|
||||
metric = "accuracy"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 79.9
|
||||
metric = "accuracy"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "IFEval"
|
||||
score = 90.0
|
||||
metric = "accuracy"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FORTE"
|
||||
score = 73.2
|
||||
metric = "success rate"
|
||||
source = "https://github.com/meituan-longcat/longcat-2.0"
|
||||
date = "2026-06-30"
|
||||
|
||||
@@ -17,3 +17,84 @@ output = 32_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf", "video"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 61.5
|
||||
metric = "resolve rate"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 80.0
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "DeepSWE"
|
||||
score = 53.3
|
||||
metric = "resolve rate"
|
||||
version = "1.1"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 88.1
|
||||
metric = "success rate"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "JobBench"
|
||||
score = 54.7
|
||||
metric = "success rate"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon-Verified"
|
||||
score = 75.6
|
||||
metric = "success rate"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 62.1
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 80.8
|
||||
metric = "success rate"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Finance Agent"
|
||||
score = 57.2
|
||||
metric = "accuracy"
|
||||
version = "v2"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "CharXiv Reasoning"
|
||||
score = 88.4
|
||||
metric = "accuracy"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BabyVision"
|
||||
score = 76.3
|
||||
metric = "accuracy"
|
||||
source = "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
|
||||
date = "2026-07-09"
|
||||
|
||||
@@ -28,3 +28,30 @@ type = "model_card"
|
||||
label = "Announcement"
|
||||
url = "https://microsoft.ai/news/introducingmai-code-1-flash/"
|
||||
type = "announcement"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 51.2
|
||||
metric = "resolve rate"
|
||||
harness = "GitHub Copilot"
|
||||
source = "https://microsoft.ai/news/introducingmai-code-1-flash/"
|
||||
date = "2026-06-02"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 71.6
|
||||
metric = "resolved"
|
||||
source = "https://llm-stats.com/benchmarks/swe-bench-verified"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 54.8
|
||||
metric = "success rate"
|
||||
version = "2.0"
|
||||
source = "https://llm-stats.com/benchmarks/terminal-bench-2"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 84.6
|
||||
metric = "accuracy"
|
||||
source = "https://llm-stats.com/benchmarks/gpqa"
|
||||
|
||||
@@ -20,3 +20,27 @@ output = ["text"]
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/MiniMaxAI/MiniMax-M2.7"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 79.9
|
||||
metric = "resolved"
|
||||
harness = "Claude Code"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 56.2
|
||||
metric = "resolve rate"
|
||||
harness = "Claude Code"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 51.1
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
@@ -20,3 +20,48 @@ output = ["text"]
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/MiniMaxAI/MiniMax-M3"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 80.5
|
||||
metric = "resolved"
|
||||
harness = "Claude Code"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 59.0
|
||||
metric = "resolve rate"
|
||||
harness = "Claude Code"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 66.0
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 83.52
|
||||
metric = "accuracy"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 74.2
|
||||
metric = "success rate"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 70.06
|
||||
metric = "success rate"
|
||||
source = "https://www.minimax.io/blog/minimax-m3"
|
||||
date = "2026-06-01"
|
||||
|
||||
@@ -3,7 +3,7 @@ description = "Earlier Kimi frontier model for long-context agents, coding, and
|
||||
family = "kimi-k2"
|
||||
release_date = "2026-01"
|
||||
last_updated = "2026-01"
|
||||
attachment = false
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
|
||||
@@ -22,3 +22,48 @@ output = ["text"]
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Kimi Code Bench"
|
||||
score = 62.0
|
||||
harness = "Kimi Code CLI"
|
||||
version = "v2"
|
||||
source = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
date = "2026-06-12"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Program Bench"
|
||||
score = 53.6
|
||||
harness = "Kimi Code CLI"
|
||||
source = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
date = "2026-06-12"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MLS Bench Lite"
|
||||
score = 35.1
|
||||
harness = "Kimi Code CLI"
|
||||
source = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
date = "2026-06-12"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 76.0
|
||||
metric = "success rate"
|
||||
harness = "Kimi Code CLI"
|
||||
source = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
date = "2026-06-12"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Mark Verified"
|
||||
score = 81.1
|
||||
metric = "success rate"
|
||||
harness = "Kimi Code CLI"
|
||||
source = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
date = "2026-06-12"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Kimi Claw 24/7 Bench"
|
||||
score = 46.9
|
||||
harness = "Kimi Code CLI"
|
||||
source = "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
|
||||
date = "2026-06-12"
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
name = "Kimi K3"
|
||||
description = "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work"
|
||||
family = "kimi-k3"
|
||||
release_date = "2026-07-16"
|
||||
last_updated = "2026-07-16"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 131_072
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
@@ -16,3 +16,86 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 70.7
|
||||
metric = "resolved"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Multilingual"
|
||||
score = 67.7
|
||||
metric = "resolve rate"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 56.4
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA"
|
||||
score = 87.0
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 26.7
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 37.4
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "LiveCodeBench"
|
||||
score = 89.0
|
||||
metric = "pass@1"
|
||||
version = "v6"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMLU-Pro"
|
||||
score = 86.8
|
||||
metric = "accuracy"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 44.4
|
||||
metric = "accuracy"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "IFBench"
|
||||
score = 81.7
|
||||
metric = "accuracy"
|
||||
variant = "prompt loose"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval"
|
||||
score = 46.7
|
||||
metric = "wins or ties"
|
||||
source = "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
|
||||
date = "2026-06-04"
|
||||
|
||||
@@ -19,3 +19,87 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 94.4
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 42.7
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 58.7
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 89.3
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval"
|
||||
score = 82.0
|
||||
metric = "wins or ties"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 50.0
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 38.0
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 4"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ARC-AGI-1"
|
||||
score = 94.5
|
||||
metric = "accuracy"
|
||||
variant = "Verified"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ARC-AGI-2"
|
||||
score = 83.3
|
||||
metric = "accuracy"
|
||||
variant = "Verified"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FinanceAgent"
|
||||
score = 61.5
|
||||
metric = "accuracy"
|
||||
version = "1.1"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GeneBench"
|
||||
score = 25.6
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
@@ -129,3 +129,87 @@ harness = "Cursor CLI"
|
||||
variant = "medium"
|
||||
version = "2.1"
|
||||
source = "https://artificialanalysis.ai/agents/coding-agents"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 75.1
|
||||
metric = "success rate"
|
||||
version = "2.0"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 92.8
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 39.8
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 52.1
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 75.0
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 82.7
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval"
|
||||
score = 83.0
|
||||
metric = "wins or ties"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ARC-AGI-2"
|
||||
score = 73.3
|
||||
metric = "accuracy"
|
||||
variant = "Verified"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 47.6
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 27.1
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 4"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 81.2
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
@@ -19,3 +19,56 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 90.1
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 43.1
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 57.2
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 52.4
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 39.6
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 4"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval"
|
||||
score = 82.3
|
||||
metric = "wins or ties"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GeneBench"
|
||||
score = 33.2
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
@@ -158,3 +158,109 @@ harness = "Cursor CLI"
|
||||
variant = "medium"
|
||||
version = "2.1"
|
||||
source = "https://artificialanalysis.ai/agents/coding-agents"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 82.7
|
||||
metric = "success rate"
|
||||
version = "2.0"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 93.6
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 41.4
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 52.2
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld-Verified"
|
||||
score = 78.7
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 84.4
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 81.2
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ARC-AGI-2"
|
||||
score = 85.0
|
||||
metric = "accuracy"
|
||||
variant = "Verified"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 51.7
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 35.4
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 4"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval"
|
||||
score = 84.9
|
||||
metric = "wins or ties"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MCP Atlas"
|
||||
score = 75.3
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon"
|
||||
score = 55.6
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "τ²-Bench Telecom"
|
||||
score = 98.0
|
||||
metric = "success rate"
|
||||
variant = "original prompts"
|
||||
source = "https://openai.com/index/introducing-gpt-5-5/"
|
||||
date = "2026-04-23"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name = "GPT-5.6 Luna"
|
||||
description = "Cost-efficient GPT-5.6 model for fast, high-volume workloads"
|
||||
family = "gpt-nano"
|
||||
family = "gpt-luna"
|
||||
release_date = "2026-07-09"
|
||||
last_updated = "2026-07-09"
|
||||
attachment = true
|
||||
@@ -19,3 +19,97 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 62.7
|
||||
metric = "resolve rate"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 84.7
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "DeepSWE"
|
||||
score = 67.2
|
||||
metric = "resolve rate"
|
||||
version = "1.1"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 92.3
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 78.6
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
version = "v2"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 83.3
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld"
|
||||
score = 45.6
|
||||
metric = "success rate"
|
||||
version = "2.0"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 78.4
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Agents' Last Exam"
|
||||
score = 50.3
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon"
|
||||
score = 53.4
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Intelligence Index"
|
||||
score = 51.2
|
||||
metric = "index score"
|
||||
variant = "max"
|
||||
version = "4.1"
|
||||
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Coding Agent Index"
|
||||
score = 74.6
|
||||
metric = "index score"
|
||||
harness = "Codex"
|
||||
variant = "max"
|
||||
version = "1.1"
|
||||
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
|
||||
date = "2026-07-09"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name = "GPT-5.6 Sol"
|
||||
description = "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows"
|
||||
family = "gpt"
|
||||
family = "gpt-sol"
|
||||
release_date = "2026-07-09"
|
||||
last_updated = "2026-07-09"
|
||||
attachment = true
|
||||
@@ -19,3 +19,97 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 64.6
|
||||
metric = "resolve rate"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 88.8
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "DeepSWE"
|
||||
score = 72.7
|
||||
metric = "resolve rate"
|
||||
version = "1.1"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 94.6
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 89
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
version = "v2"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 90.4
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld"
|
||||
score = 62.6
|
||||
metric = "success rate"
|
||||
version = "2.0"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 83
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Agents' Last Exam"
|
||||
score = 52.7
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon"
|
||||
score = 58
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Intelligence Index"
|
||||
score = 58.9
|
||||
metric = "index score"
|
||||
variant = "max"
|
||||
version = "4.1"
|
||||
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Coding Agent Index"
|
||||
score = 80
|
||||
metric = "index score"
|
||||
harness = "Codex"
|
||||
variant = "max"
|
||||
version = "1.1"
|
||||
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
|
||||
date = "2026-07-09"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name = "GPT-5.6 Terra"
|
||||
description = "Balanced GPT-5.6 model for capable, cost-efficient everyday work"
|
||||
family = "gpt-mini"
|
||||
family = "gpt-terra"
|
||||
release_date = "2026-07-09"
|
||||
last_updated = "2026-07-09"
|
||||
attachment = true
|
||||
@@ -19,3 +19,97 @@ output = 128_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 63.4
|
||||
metric = "resolve rate"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 87.4
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "DeepSWE"
|
||||
score = 69.6
|
||||
metric = "resolve rate"
|
||||
version = "1.1"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 92.9
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierMath"
|
||||
score = 84.9
|
||||
metric = "accuracy"
|
||||
dataset = "Tier 1-3"
|
||||
version = "v2"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 87.5
|
||||
metric = "accuracy"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "OSWorld"
|
||||
score = 50.2
|
||||
metric = "success rate"
|
||||
version = "2.0"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "MMMU Pro"
|
||||
score = 80.7
|
||||
metric = "accuracy"
|
||||
variant = "no tools"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Agents' Last Exam"
|
||||
score = 50.4
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon"
|
||||
score = 53.1
|
||||
metric = "success rate"
|
||||
source = "https://openai.com/index/gpt-5-6/"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Intelligence Index"
|
||||
score = 55
|
||||
metric = "index score"
|
||||
variant = "max"
|
||||
version = "4.1"
|
||||
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
|
||||
date = "2026-07-09"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Coding Agent Index"
|
||||
score = 77.4
|
||||
metric = "index score"
|
||||
harness = "Codex"
|
||||
variant = "max"
|
||||
version = "1.1"
|
||||
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
|
||||
date = "2026-07-09"
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
name = "GPT-Realtime-2.1"
|
||||
description = "Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior"
|
||||
family = "gpt"
|
||||
release_date = "2026-07-06"
|
||||
last_updated = "2026-07-06"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = false
|
||||
tool_call = true
|
||||
structured_output = false
|
||||
knowledge = "2024-09-30"
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 128_000
|
||||
input = 96_000
|
||||
output = 32_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "audio", "image"]
|
||||
output = ["text", "audio"]
|
||||
@@ -0,0 +1,18 @@
|
||||
name = "GPT Realtime Whisper"
|
||||
description = "Streaming speech-to-text model for low-latency transcript deltas from live audio"
|
||||
family = "whisper"
|
||||
release_date = "2026-05-07"
|
||||
last_updated = "2026-05-07"
|
||||
attachment = false
|
||||
reasoning = false
|
||||
temperature = true
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 0
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["audio"]
|
||||
output = ["text"]
|
||||
@@ -1,5 +1,5 @@
|
||||
name = "Laguna M.1"
|
||||
description = "Poolside's flagship agentic coding model for long-horizon work"
|
||||
description = "Poolside's open-weight model for agentic coding and long-horizon work"
|
||||
family = "laguna"
|
||||
release_date = "2026-04-28"
|
||||
last_updated = "2026-06-13"
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
name = "Laguna S 2.1"
|
||||
description = "Agentic coding model from Poolside in the XS size class for local deployment"
|
||||
family = "laguna"
|
||||
release_date = "2026-07-21"
|
||||
last_updated = "2026-07-21"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = false
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 32_768
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
@@ -17,3 +17,36 @@ output = 32_768
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 70.9
|
||||
metric = "resolved"
|
||||
harness = "Harbor"
|
||||
source = "https://poolside.ai/blog/introducing-laguna-xs-2-1"
|
||||
date = "2026-07-02"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Multilingual"
|
||||
score = 63.1
|
||||
metric = "resolve rate"
|
||||
harness = "Harbor"
|
||||
source = "https://poolside.ai/blog/introducing-laguna-xs-2-1"
|
||||
date = "2026-07-02"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 47.6
|
||||
metric = "resolve rate"
|
||||
harness = "Harbor"
|
||||
source = "https://poolside.ai/blog/introducing-laguna-xs-2-1"
|
||||
date = "2026-07-02"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 37.5
|
||||
metric = "success rate"
|
||||
harness = "Harbor"
|
||||
version = "2.0"
|
||||
source = "https://poolside.ai/blog/introducing-laguna-xs-2-1"
|
||||
date = "2026-07-02"
|
||||
|
||||
@@ -6,7 +6,7 @@ attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
knowledge = "2026-01-01"
|
||||
knowledge = "2026-03-01"
|
||||
open_weights = true
|
||||
|
||||
[limit]
|
||||
@@ -15,9 +15,89 @@ input = 256_000
|
||||
output = 256_000
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/stepfun-ai/Step-3.7-Flash"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 56.3
|
||||
metric = "resolve rate"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 76.5
|
||||
metric = "resolved"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 59.6
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Humanity's Last Exam"
|
||||
score = 47.2
|
||||
metric = "accuracy"
|
||||
variant = "with tools"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "BrowseComp"
|
||||
score = 75.8
|
||||
metric = "accuracy"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Toolathlon"
|
||||
score = 49.5
|
||||
metric = "success rate"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval"
|
||||
score = 45.8
|
||||
metric = "wins or ties"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "ClawEval"
|
||||
score = 67.1
|
||||
metric = "pass^3"
|
||||
version = "1.1"
|
||||
source = "https://static.stepfun.com/blog/step-3.7-flash/"
|
||||
date = "2026-05-29"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Coding Index"
|
||||
score = 37.1
|
||||
metric = "index"
|
||||
source = "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks"
|
||||
date = "2026-06-15"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SciCode"
|
||||
score = 40.0
|
||||
metric = "percent correct"
|
||||
source = "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks"
|
||||
date = "2026-06-15"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench Hard"
|
||||
score = 35.6
|
||||
metric = "success rate"
|
||||
source = "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks"
|
||||
date = "2026-06-15"
|
||||
|
||||
@@ -1,27 +1,28 @@
|
||||
name = "Hy3 (free)"
|
||||
name = "Hy3"
|
||||
description = "Tencent Hy reasoning model for coding, instruction following, and agent tasks"
|
||||
family = "hy3"
|
||||
family = "Hy"
|
||||
release_date = "2026-07-06"
|
||||
last_updated = "2026-07-06"
|
||||
attachment = false
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
structured_output = true
|
||||
open_weights = true
|
||||
|
||||
[[reasoning_options]]
|
||||
type = "effort"
|
||||
values = ["none", "low", "high"]
|
||||
|
||||
[cost]
|
||||
input = 0
|
||||
output = 0
|
||||
|
||||
[limit]
|
||||
context = 262_144
|
||||
output = 262_144
|
||||
context = 256_000
|
||||
output = 64_000
|
||||
|
||||
[modalities]
|
||||
input = ["text"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/tencent/Hy3"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Verified"
|
||||
score = 78
|
||||
metric = "resolved"
|
||||
source = "https://huggingface.co/tencent/Hy3"
|
||||
@@ -0,0 +1,28 @@
|
||||
# Sources (accessed 2026-07-22):
|
||||
# - https://thinkingmachines.ai/news/introducing-inkling/
|
||||
# - https://thinkingmachines.ai/model-card/inkling/
|
||||
# - https://huggingface.co/thinkingmachines/Inkling
|
||||
|
||||
name = "Inkling"
|
||||
description = "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio"
|
||||
family = "ling"
|
||||
release_date = "2026-07-15"
|
||||
last_updated = "2026-07-15"
|
||||
attachment = true
|
||||
reasoning = true
|
||||
temperature = true
|
||||
tool_call = true
|
||||
open_weights = true
|
||||
license = "Apache-2.0"
|
||||
|
||||
[limit]
|
||||
context = 1_048_576
|
||||
output = 1_048_576
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "audio"]
|
||||
output = ["text"]
|
||||
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/thinkingmachines/Inkling"
|
||||
@@ -17,3 +17,32 @@ output = 30_000
|
||||
[modalities]
|
||||
input = ["text", "image", "pdf"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Artificial Analysis Intelligence Index"
|
||||
score = 53
|
||||
metric = "index score"
|
||||
version = "4.0"
|
||||
source = "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing"
|
||||
date = "2026-04-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GDPval-AA"
|
||||
score = 1500
|
||||
metric = "Elo"
|
||||
source = "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing"
|
||||
date = "2026-04-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "τ²-Bench Telecom"
|
||||
score = 98
|
||||
metric = "success rate"
|
||||
source = "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing"
|
||||
date = "2026-04-30"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "IFBench"
|
||||
score = 81
|
||||
metric = "accuracy"
|
||||
source = "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing"
|
||||
date = "2026-04-30"
|
||||
|
||||
@@ -17,3 +17,42 @@ output = 500_000
|
||||
[modalities]
|
||||
input = ["text", "image"]
|
||||
output = ["text"]
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 64.7
|
||||
metric = "resolve rate"
|
||||
source = "https://x.ai/news/grok-4-5"
|
||||
date = "2026-07-08"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Multilingual"
|
||||
score = 78
|
||||
metric = "resolve rate"
|
||||
source = "https://x.ai/news/grok-4-5"
|
||||
date = "2026-07-08"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 83.3
|
||||
metric = "success rate"
|
||||
version = "2.1"
|
||||
source = "https://x.ai/news/grok-4-5"
|
||||
date = "2026-07-08"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "DeepSWE"
|
||||
score = 62.0
|
||||
metric = "resolve rate"
|
||||
version = "1.0"
|
||||
source = "https://x.ai/news/grok-4-5"
|
||||
date = "2026-07-08"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "DeepSWE"
|
||||
score = 53
|
||||
metric = "resolve rate"
|
||||
harness = "mini-swe-agent"
|
||||
version = "1.1"
|
||||
source = "https://x.ai/news/grok-4-5"
|
||||
date = "2026-07-08"
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
# Sources:
|
||||
# - https://docs.x.ai/docs/models
|
||||
# - https://docs.x.ai/developers/models/grok-imagine-video-1.5
|
||||
# - https://docs.x.ai/docs/guides/video-generation
|
||||
|
||||
name = "Grok Imagine Video 1.5"
|
||||
description = "Video model for image-to-video generation, editing, and extension workflows"
|
||||
family = "grok"
|
||||
release_date = "2026-05-30"
|
||||
last_updated = "2026-05-30"
|
||||
attachment = true
|
||||
reasoning = false
|
||||
temperature = false
|
||||
tool_call = false
|
||||
open_weights = false
|
||||
|
||||
[limit]
|
||||
context = 1_024
|
||||
output = 0
|
||||
|
||||
[modalities]
|
||||
input = ["text", "image", "video"]
|
||||
output = ["video"]
|
||||
@@ -27,3 +27,17 @@ name = "SWE-Bench Verified"
|
||||
score = 78.9
|
||||
metric = "resolved"
|
||||
source = "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 57.2
|
||||
metric = "resolve rate"
|
||||
source = "https://mimo.xiaomi.com/mimo-v2-5-pro/"
|
||||
date = "2026-04-22"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "GPQA Diamond"
|
||||
score = 86.6
|
||||
metric = "accuracy"
|
||||
source = "https://mimo.xiaomi.com/mimo-v2-5-pro/"
|
||||
date = "2026-04-22"
|
||||
|
||||
@@ -21,3 +21,26 @@ output = ["text"]
|
||||
[[weights]]
|
||||
label = "Hugging Face"
|
||||
url = "https://huggingface.co/zai-org/GLM-5.2"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "SWE-Bench Pro"
|
||||
score = 62.1
|
||||
metric = "resolve rate"
|
||||
source = "https://z.ai/blog/glm-5.2"
|
||||
date = "2026-06-16"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "Terminal-Bench"
|
||||
score = 82.7
|
||||
metric = "success rate"
|
||||
harness = "Claude Code"
|
||||
version = "2.1"
|
||||
source = "https://z.ai/blog/glm-5.2"
|
||||
date = "2026-06-16"
|
||||
|
||||
[[benchmarks]]
|
||||
name = "FrontierSWE"
|
||||
score = 74.4
|
||||
metric = "dominance"
|
||||
source = "https://z.ai/blog/glm-5.2"
|
||||
date = "2026-06-16"
|
||||
|
||||
+4
-1
@@ -26,12 +26,15 @@
|
||||
"databricks:generate": "bun ./packages/core/script/generate-databricks.ts",
|
||||
"helicone:generate": "bun ./packages/core/script/generate-helicone.ts",
|
||||
"huggingface:sync": "bun ./packages/core/script/sync-models.ts huggingface",
|
||||
"kilo:sync": "bun ./packages/core/script/sync-models.ts kilo",
|
||||
"llmgateway:sync": "bun ./packages/core/script/sync-models.ts llmgateway",
|
||||
"merge-gateway:sync": "bun ./packages/core/script/sync-models.ts merge-gateway",
|
||||
"nano-gpt:sync": "bun ./packages/core/script/sync-models.ts nano-gpt",
|
||||
"venice:sync": "bun ./packages/core/script/sync-models.ts venice",
|
||||
"vercel:generate": "bun ./packages/core/script/sync-models.ts vercel",
|
||||
"wandb:generate": "bun ./packages/core/script/sync-models.ts wandb",
|
||||
"digitalocean:sync": "bun ./packages/core/script/sync-models.ts digitalocean",
|
||||
"ambient:generate": "bun ./packages/core/script/generate-ambient.ts",
|
||||
"ambient:sync": "bun ./packages/core/script/sync-models.ts ambient",
|
||||
"models:sync": "bun ./packages/core/script/sync-models.ts",
|
||||
"sync:models": "bun ./packages/core/script/sync-models.ts"
|
||||
},
|
||||
|
||||
@@ -1,169 +0,0 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
/**
|
||||
* Generates Ambient model TOML files from https://api.ambient.xyz/v1/models.
|
||||
*
|
||||
* Emits `base_model` TOMLs that inherit upstream metadata
|
||||
* (family, release_date, knowledge, capabilities) from the canonical
|
||||
* provider model, and override only the fields Ambient's API reports:
|
||||
* cost, limit, modalities.
|
||||
*
|
||||
* Flags:
|
||||
* --dry-run Preview generated TOMLs without writing files.
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import path from "node:path";
|
||||
import { mkdir } from "node:fs/promises";
|
||||
|
||||
const API_ENDPOINT = "https://api.ambient.xyz/v1/models";
|
||||
|
||||
// Allowlist for the initial rollout.
|
||||
const ALLOWLIST = new Set<string>([
|
||||
"zai-org/GLM-5.1-FP8",
|
||||
"moonshotai/kimi-k2.6",
|
||||
]);
|
||||
|
||||
// Maps Ambient model IDs to canonical model metadata IDs in this repo.
|
||||
const BASE_MODEL_MAP: Record<string, string> = {
|
||||
"zai-org/GLM-5.1-FP8": "zhipuai/glm-5.1",
|
||||
"moonshotai/kimi-k2.6": "moonshotai/kimi-k2.6",
|
||||
};
|
||||
|
||||
const Pricing = z
|
||||
.object({
|
||||
prompt: z.string(),
|
||||
completion: z.string(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const AmbientModel = z
|
||||
.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
context_length: z.number(),
|
||||
max_output_length: z.number(),
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
pricing: Pricing,
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const AmbientResponse = z
|
||||
.object({
|
||||
object: z.literal("list"),
|
||||
data: z.array(AmbientModel),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const ALLOWED_MODALITIES = new Set(["text", "audio", "image", "video", "pdf"]);
|
||||
|
||||
function modalities(values: string[]): string[] {
|
||||
return values
|
||||
.map((v) => v.toLowerCase())
|
||||
.filter((v) => ALLOWED_MODALITIES.has(v));
|
||||
}
|
||||
|
||||
function perMTok(price: string): number {
|
||||
const n = parseFloat(price);
|
||||
if (!Number.isFinite(n)) {
|
||||
throw new Error(`Invalid price: ${price}`);
|
||||
}
|
||||
// Round to 6 decimals to absorb float noise from per-token strings.
|
||||
return Math.round(n * 1_000_000 * 1_000_000) / 1_000_000;
|
||||
}
|
||||
|
||||
function formatToml(
|
||||
model: z.infer<typeof AmbientModel>,
|
||||
baseModel: string,
|
||||
): string {
|
||||
const lines: string[] = [];
|
||||
lines.push(`base_model = "${baseModel}"`);
|
||||
lines.push("");
|
||||
|
||||
lines.push("[cost]");
|
||||
lines.push(`input = ${perMTok(model.pricing.prompt)}`);
|
||||
lines.push(`output = ${perMTok(model.pricing.completion)}`);
|
||||
if (model.pricing.input_cache_read !== undefined) {
|
||||
lines.push(`cache_read = ${perMTok(model.pricing.input_cache_read)}`);
|
||||
}
|
||||
if (model.pricing.input_cache_write !== undefined) {
|
||||
lines.push(`cache_write = ${perMTok(model.pricing.input_cache_write)}`);
|
||||
}
|
||||
lines.push("");
|
||||
|
||||
lines.push("[limit]");
|
||||
lines.push(`context = ${model.context_length}`);
|
||||
lines.push(`output = ${model.max_output_length}`);
|
||||
lines.push("");
|
||||
|
||||
const input = modalities(model.input_modalities);
|
||||
const output = modalities(model.output_modalities);
|
||||
lines.push("[modalities]");
|
||||
lines.push(`input = [${input.map((m) => `"${m}"`).join(", ")}]`);
|
||||
lines.push(`output = [${output.map((m) => `"${m}"`).join(", ")}]`);
|
||||
|
||||
return lines.join("\n") + "\n";
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const dryRun = process.argv.includes("--dry-run");
|
||||
|
||||
const outDir = path.join(
|
||||
import.meta.dirname,
|
||||
"..",
|
||||
"..",
|
||||
"..",
|
||||
"providers",
|
||||
"ambient",
|
||||
"models",
|
||||
);
|
||||
|
||||
const res = await fetch(API_ENDPOINT);
|
||||
if (!res.ok) {
|
||||
console.error(`Fetch failed: ${res.status} ${res.statusText}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const parsed = AmbientResponse.safeParse(await res.json());
|
||||
if (!parsed.success) {
|
||||
console.error("Invalid Ambient response:", parsed.error.issues);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const selected = parsed.data.data.filter((m) => ALLOWLIST.has(m.id));
|
||||
const missing = [...ALLOWLIST].filter(
|
||||
(id) => !selected.some((m) => m.id === id),
|
||||
);
|
||||
if (missing.length > 0) {
|
||||
console.error(`Allowlisted models missing from API: ${missing.join(", ")}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
let count = 0;
|
||||
for (const model of selected) {
|
||||
const baseModel = BASE_MODEL_MAP[model.id];
|
||||
if (!baseModel) {
|
||||
console.error(`No BASE_MODEL_MAP entry for ${model.id}; skipping`);
|
||||
continue;
|
||||
}
|
||||
const filePath = path.join(outDir, `${model.id}.toml`);
|
||||
const toml = formatToml(model, baseModel);
|
||||
if (dryRun) {
|
||||
console.log(`--- ${path.relative(process.cwd(), filePath)} ---`);
|
||||
console.log(toml);
|
||||
} else {
|
||||
await mkdir(path.dirname(filePath), { recursive: true });
|
||||
await Bun.write(filePath, toml);
|
||||
}
|
||||
count++;
|
||||
}
|
||||
|
||||
console.log(
|
||||
`${dryRun ? "Previewed" : "Wrote"} ${count} model file(s) under providers/ambient/models/`,
|
||||
);
|
||||
}
|
||||
|
||||
await main();
|
||||
@@ -13,6 +13,9 @@ export const ModelFamilyValues = [
|
||||
"gpt-pro",
|
||||
"gpt-mini",
|
||||
"gpt-nano",
|
||||
"gpt-sol",
|
||||
"gpt-terra",
|
||||
"gpt-luna",
|
||||
"gpt-oss",
|
||||
"gpt-image",
|
||||
|
||||
@@ -73,11 +76,13 @@ export const ModelFamilyValues = [
|
||||
// Moonshot Kimi
|
||||
"kimi",
|
||||
"kimi-k2",
|
||||
"kimi-k3",
|
||||
"kimi-free",
|
||||
"kimi-thinking",
|
||||
|
||||
// Poolside Laguna
|
||||
"laguna",
|
||||
"laguna-s",
|
||||
|
||||
// Mistral family
|
||||
"mistral",
|
||||
@@ -442,5 +447,6 @@ export function inferKimiFamily(...values: string[]): ModelFamily | undefined {
|
||||
const target = values.join(" ").toLowerCase();
|
||||
if (/kimi[^a-z0-9]*k2(?:[^a-z0-9]*\d+)?[^a-z0-9]*thinking/.test(target)) return "kimi-thinking";
|
||||
if (/kimi[\s_-]*k2/.test(target)) return "kimi-k2";
|
||||
if (/kimi[\s_-]*k3/.test(target)) return "kimi-k3";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
@@ -4,6 +4,8 @@ import { mergeDeep } from "remeda";
|
||||
import { z } from "zod";
|
||||
|
||||
import { AuthoredModel, AuthoredModelShape, ModelMetadata } from "../schema.js";
|
||||
import { openMissingModelIssues } from "./missing-issues.js";
|
||||
import { ambient } from "./providers/ambient.js";
|
||||
import { anthropic } from "./providers/anthropic.js";
|
||||
import { baseten } from "./providers/baseten.js";
|
||||
import { chutes } from "./providers/chutes.js";
|
||||
@@ -11,12 +13,18 @@ import { cloudflareWorkersAi } from "./providers/cloudflare-workers-ai.js";
|
||||
import { crossmodel } from "./providers/crossmodel.js";
|
||||
import { deepinfra } from "./providers/deepinfra.js";
|
||||
import { digitalocean } from "./providers/digitalocean.js";
|
||||
import { empiriolabs } from "./providers/empiriolabs.js";
|
||||
import { google } from "./providers/google.js";
|
||||
import { hyper } from "./providers/hyper.js";
|
||||
import { huggingface } from "./providers/huggingface.js";
|
||||
import { kilo } from "./providers/kilo.js";
|
||||
import { llmgateway } from "./providers/llmgateway.js";
|
||||
import { mergeGateway } from "./providers/merge-gateway.js";
|
||||
import { nanoGpt } from "./providers/nano-gpt.js";
|
||||
import { openai } from "./providers/openai.js";
|
||||
import { openrouter } from "./providers/openrouter.js";
|
||||
import { ovhcloud } from "./providers/ovhcloud.js";
|
||||
import { pioneer } from "./providers/pioneer.js";
|
||||
import { vercel } from "./providers/vercel.js";
|
||||
import { venice } from "./providers/venice.js";
|
||||
import { wandb } from "./providers/wandb.js";
|
||||
@@ -57,13 +65,24 @@ export interface SyncProvider<SourceModel> {
|
||||
name: string;
|
||||
modelsDir: string;
|
||||
metadataNamespace?: string;
|
||||
/**
|
||||
* Do not create new local TOMLs for remote-only models. Instead open one
|
||||
* deduped GitHub issue per missing model ID.
|
||||
*/
|
||||
skipCreates?: boolean;
|
||||
/** Report remote-only models skipped by skipCreates as GitHub issues. */
|
||||
trackMissingModels?: boolean;
|
||||
deleteMissing?: boolean;
|
||||
preserveSymlinks?: boolean;
|
||||
preserveBaseModels?: boolean;
|
||||
preserveDescriptions?: boolean;
|
||||
sameModel?(current: ExistingModel, desired: SyncedModel): boolean;
|
||||
missingNotice?(paths: string[]): string[];
|
||||
sourceID?(model: SourceModel): string;
|
||||
/**
|
||||
* Remote ID to report when translateModel skips a source model. Return
|
||||
* undefined to skip silently (no notice, no missing-model issue).
|
||||
*/
|
||||
sourceID?(model: SourceModel): string | undefined;
|
||||
skippedNotice?(ids: string[]): string[];
|
||||
fetchModels(): Promise<unknown>;
|
||||
parseModels(raw: unknown): SourceModel[];
|
||||
@@ -89,6 +108,7 @@ export interface SyncResult {
|
||||
}
|
||||
|
||||
export const providers: {
|
||||
ambient: SyncProvider<any>;
|
||||
anthropic: SyncProvider<any>;
|
||||
baseten: SyncProvider<any>;
|
||||
chutes: SyncProvider<any>;
|
||||
@@ -96,17 +116,24 @@ export const providers: {
|
||||
crossmodel: SyncProvider<any>;
|
||||
deepinfra: SyncProvider<any>;
|
||||
digitalocean: SyncProvider<any>;
|
||||
empiriolabs: SyncProvider<any>;
|
||||
google: SyncProvider<any>;
|
||||
hyper: SyncProvider<any>;
|
||||
huggingface: SyncProvider<any>;
|
||||
kilo: SyncProvider<any>;
|
||||
llmgateway: SyncProvider<any>;
|
||||
"merge-gateway": SyncProvider<any>;
|
||||
"nano-gpt": SyncProvider<any>;
|
||||
openai: SyncProvider<any>;
|
||||
openrouter: SyncProvider<any>;
|
||||
ovhcloud: SyncProvider<any>;
|
||||
pioneer: SyncProvider<any>;
|
||||
vercel: SyncProvider<any>;
|
||||
venice: SyncProvider<any>;
|
||||
wandb: SyncProvider<any>;
|
||||
xai: SyncProvider<any>;
|
||||
} = {
|
||||
ambient,
|
||||
anthropic,
|
||||
baseten,
|
||||
chutes,
|
||||
@@ -114,12 +141,18 @@ export const providers: {
|
||||
crossmodel,
|
||||
deepinfra,
|
||||
digitalocean,
|
||||
empiriolabs,
|
||||
google,
|
||||
hyper,
|
||||
huggingface,
|
||||
kilo,
|
||||
llmgateway,
|
||||
"merge-gateway": mergeGateway,
|
||||
"nano-gpt": nanoGpt,
|
||||
openai,
|
||||
openrouter,
|
||||
ovhcloud,
|
||||
pioneer,
|
||||
vercel,
|
||||
venice,
|
||||
wandb,
|
||||
@@ -127,15 +160,26 @@ export const providers: {
|
||||
};
|
||||
|
||||
export const groups = {
|
||||
aggregators: ["crossmodel", "huggingface", "llmgateway", "openrouter", "vercel"],
|
||||
aggregators: [
|
||||
"crossmodel",
|
||||
"empiriolabs",
|
||||
"huggingface",
|
||||
"kilo",
|
||||
"llmgateway",
|
||||
"merge-gateway",
|
||||
"nano-gpt",
|
||||
"openrouter",
|
||||
"vercel",
|
||||
],
|
||||
cloudflare: ["cloudflare-workers-ai"],
|
||||
direct: ["anthropic", "baseten", "chutes", "deepinfra", "digitalocean", "google", "openai", "ovhcloud", "venice", "wandb", "xai"],
|
||||
direct: ["ambient", "anthropic", "baseten", "chutes", "deepinfra", "digitalocean", "google", "hyper", "openai", "ovhcloud", "pioneer", "venice", "wandb", "xai"],
|
||||
} as const;
|
||||
|
||||
type ProviderID = keyof typeof providers;
|
||||
|
||||
interface SyncOptions {
|
||||
dryRun?: boolean;
|
||||
openIssues?: boolean;
|
||||
newOnly?: boolean;
|
||||
}
|
||||
|
||||
@@ -167,12 +211,13 @@ export async function syncProvider<SourceModel>(
|
||||
},
|
||||
});
|
||||
if (translated === undefined) {
|
||||
if (provider.sourceID !== undefined) skippedRemote.push(provider.sourceID(sourceModel));
|
||||
const skippedID = provider.sourceID?.(sourceModel);
|
||||
if (skippedID !== undefined) skippedRemote.push(skippedID);
|
||||
continue;
|
||||
}
|
||||
|
||||
const relativePath = `${translated.id}.toml`;
|
||||
if (provider.skipCreates && !existing.has(relativePath)) {
|
||||
if (provider.skipCreates === true && !existing.has(relativePath)) {
|
||||
skippedRemote.push(translated.id);
|
||||
continue;
|
||||
}
|
||||
@@ -214,16 +259,17 @@ export async function syncProvider<SourceModel>(
|
||||
} else {
|
||||
resolvedReasoning = existing.get(relativePath)?.toml.reasoning;
|
||||
}
|
||||
const withReasoningOptions = preserveReasoningOptions(
|
||||
translatedModel,
|
||||
existing.get(relativePath)?.authored,
|
||||
resolvedReasoning,
|
||||
);
|
||||
const withDescription = provider.preserveDescriptions === false
|
||||
? withReasoningOptions
|
||||
: preserveDescription(withReasoningOptions, existing.get(relativePath)?.authored);
|
||||
const parsed = SyncedAuthoredModel.safeParse(stripUndefined({
|
||||
id: translated.id,
|
||||
...preserveDescription(
|
||||
preserveReasoningOptions(
|
||||
translatedModel,
|
||||
existing.get(relativePath)?.authored,
|
||||
resolvedReasoning,
|
||||
),
|
||||
existing.get(relativePath)?.authored,
|
||||
),
|
||||
...withDescription,
|
||||
}));
|
||||
if (!parsed.success) {
|
||||
parsed.error.cause = { provider: provider.id, path: relativePath };
|
||||
@@ -345,10 +391,38 @@ export async function syncProvider<SourceModel>(
|
||||
}
|
||||
}
|
||||
|
||||
const result = summarize(provider, files, unchanged, [
|
||||
const notices = [
|
||||
...provider.skippedNotice?.(skippedRemote) ?? [],
|
||||
...provider.missingNotice?.(missingLocal) ?? [],
|
||||
]);
|
||||
];
|
||||
|
||||
if (
|
||||
provider.skipCreates === true
|
||||
&& provider.trackMissingModels !== false
|
||||
&& skippedRemote.length > 0
|
||||
&& options.openIssues === true
|
||||
) {
|
||||
try {
|
||||
notices.push(
|
||||
...await openMissingModelIssues(
|
||||
{ id: provider.id, name: provider.name, modelsDir: provider.modelsDir },
|
||||
skippedRemote,
|
||||
{ dryRun: options.dryRun },
|
||||
),
|
||||
);
|
||||
} catch (error) {
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
const notice = `Failed to open missing-model GitHub issues: ${message}`;
|
||||
notices.push(notice);
|
||||
console.error(notice);
|
||||
// Surface as a workflow annotation: on no-change hours the notice never
|
||||
// reaches a PR body, so a broken token or full dedupe window would
|
||||
// otherwise disable issue opens silently while runs stay green.
|
||||
if (process.env.GITHUB_ACTIONS === "true") console.log(`::error::${provider.id}: ${notice}`);
|
||||
}
|
||||
}
|
||||
|
||||
const result = summarize(provider, files, unchanged, notices);
|
||||
console.log(
|
||||
`${options.dryRun ? "Dry run: " : ""}${result.created} created, ${result.updated} updated, ${result.deleted} removed, ${result.unchanged} unchanged`,
|
||||
);
|
||||
@@ -922,6 +996,9 @@ export async function main(args = process.argv.slice(2)) {
|
||||
const results = await syncTargets(target, {
|
||||
dryRun: args.includes("--dry-run"),
|
||||
newOnly: args.includes("--new-only"),
|
||||
// Only GitHub Actions opens issues by default; local needs --open-issues.
|
||||
openIssues: args.includes("--open-issues")
|
||||
|| (process.env.GITHUB_ACTIONS === "true" && !args.includes("--no-issues")),
|
||||
});
|
||||
|
||||
await writeReport(target, results);
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
export interface MissingModelIssueTarget {
|
||||
id: string;
|
||||
name: string;
|
||||
modelsDir: string;
|
||||
}
|
||||
|
||||
export interface OpenMissingModelIssuesOptions {
|
||||
dryRun?: boolean;
|
||||
}
|
||||
|
||||
function issueTitle(providerId: string, modelId: string) {
|
||||
return `[missing-model] ${providerId}: ${modelId}`;
|
||||
}
|
||||
|
||||
function issueBody(provider: MissingModelIssueTarget, modelId: string) {
|
||||
return [
|
||||
`The **${provider.name}** catalog sync found remote model \`${modelId}\` that is not in the local catalog.`,
|
||||
"",
|
||||
`| Field | Value |`,
|
||||
`| --- | --- |`,
|
||||
`| Provider | \`${provider.id}\` |`,
|
||||
`| Model ID | \`${modelId}\` |`,
|
||||
`| Expected path | \`${provider.modelsDir}/${modelId}.toml\` |`,
|
||||
"",
|
||||
"This provider uses `skipCreates` because the remote source is not enough to auto-author a full TOML.",
|
||||
"Add the model manually (prefer `base_model` when matching `models/` metadata exists).",
|
||||
"",
|
||||
].join("\n");
|
||||
}
|
||||
|
||||
/** Open one deduped GitHub issue per missing model ID (title-stable). */
|
||||
export async function openMissingModelIssues(
|
||||
provider: MissingModelIssueTarget,
|
||||
modelIds: string[],
|
||||
options: OpenMissingModelIssuesOptions = {},
|
||||
): Promise<string[]> {
|
||||
const ids = [...new Set(modelIds)].filter((id) => id.length > 0).sort();
|
||||
if (ids.length === 0) return [];
|
||||
|
||||
const notices: string[] = [];
|
||||
const labels = ["automation", "model-sync", "missing-model", `provider:${provider.id}`];
|
||||
|
||||
if (options.dryRun) {
|
||||
for (const modelId of ids) {
|
||||
const notice = `Would open GitHub issue for missing model \`${modelId}\` (\`${issueTitle(provider.id, modelId)}\`)`;
|
||||
notices.push(notice);
|
||||
console.log(notice);
|
||||
}
|
||||
return notices;
|
||||
}
|
||||
|
||||
// Fail closed before listing/creating: a label failure here would otherwise
|
||||
// surface as one opaque `gh issue create` error per model.
|
||||
for (const label of labels) {
|
||||
const result = await runGh([
|
||||
"label",
|
||||
"create",
|
||||
label,
|
||||
"--color",
|
||||
"0E8A16",
|
||||
"--description",
|
||||
"Automated model catalog sync",
|
||||
"--force",
|
||||
]);
|
||||
if (result.code !== 0) {
|
||||
throw new Error(`gh label create ${label} failed: ${result.stderr || result.stdout || `exit ${result.code}`}`);
|
||||
}
|
||||
}
|
||||
|
||||
const existingByTitle = await listTrackedTitles(provider.id);
|
||||
|
||||
for (const modelId of ids) {
|
||||
const title = issueTitle(provider.id, modelId);
|
||||
const existing = existingByTitle.get(title);
|
||||
if (existing !== undefined) {
|
||||
const notice = `Missing model \`${modelId}\` already tracked by #${existing}`;
|
||||
notices.push(notice);
|
||||
console.log(notice);
|
||||
continue;
|
||||
}
|
||||
|
||||
try {
|
||||
const number = await createIssue(title, issueBody(provider, modelId), labels);
|
||||
existingByTitle.set(title, number);
|
||||
await dispatchIssueFixer(provider.id, number);
|
||||
const notice = `Opened GitHub issue #${number} and dispatched the issue fixer for missing model \`${modelId}\``;
|
||||
notices.push(notice);
|
||||
console.log(notice);
|
||||
} catch (error) {
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
const notice = `Failed to open GitHub issue for missing model \`${modelId}\`: ${message}`;
|
||||
notices.push(notice);
|
||||
console.error(notice);
|
||||
}
|
||||
}
|
||||
|
||||
return notices;
|
||||
}
|
||||
|
||||
const LIST_LIMIT = 1000;
|
||||
|
||||
async function listTrackedTitles(providerId: string) {
|
||||
// Include closed so a wontfix/closed issue does not reopen hourly.
|
||||
const result = await runGh([
|
||||
"issue",
|
||||
"list",
|
||||
"--state",
|
||||
"all",
|
||||
"--label",
|
||||
"missing-model",
|
||||
"--label",
|
||||
`provider:${providerId}`,
|
||||
"--limit",
|
||||
String(LIST_LIMIT),
|
||||
"--json",
|
||||
"number,title",
|
||||
]);
|
||||
if (result.code !== 0) {
|
||||
throw new Error(`gh issue list failed: ${result.stderr || result.stdout || `exit ${result.code}`}`);
|
||||
}
|
||||
|
||||
const issues = JSON.parse(result.stdout || "[]") as Array<{ number: number; title: string }>;
|
||||
// Fail closed when the window is full: older titles may have been truncated,
|
||||
// and creating against an incomplete list could reopen duplicates.
|
||||
if (issues.length >= LIST_LIMIT) {
|
||||
throw new Error(
|
||||
`gh issue list returned ${issues.length} issues (window limit ${LIST_LIMIT}); refusing to create against a possibly truncated dedupe list`,
|
||||
);
|
||||
}
|
||||
return new Map(issues.map((issue) => [issue.title, issue.number]));
|
||||
}
|
||||
|
||||
async function createIssue(title: string, body: string, labels: string[]) {
|
||||
const args = ["issue", "create", "--title", title, "--body", body];
|
||||
for (const label of labels) args.push("--label", label);
|
||||
const result = await runGh(args);
|
||||
if (result.code !== 0) {
|
||||
throw new Error(`gh issue create failed: ${result.stderr || result.stdout || `exit ${result.code}`}`);
|
||||
}
|
||||
|
||||
const url = result.stdout.trim();
|
||||
const number = url.match(/\/issues\/(\d+)\s*$/)?.[1] ?? url.match(/#(\d+)\s*$/)?.[1];
|
||||
if (number === undefined) {
|
||||
throw new Error(`gh issue create returned no issue number: ${url}`);
|
||||
}
|
||||
return Number(number);
|
||||
}
|
||||
|
||||
async function dispatchIssueFixer(providerId: string, issueNumber: number) {
|
||||
const repository = process.env.GITHUB_REPOSITORY;
|
||||
if (repository === undefined) {
|
||||
throw new Error("GITHUB_REPOSITORY is required to dispatch the issue fixer");
|
||||
}
|
||||
|
||||
const result = await runGh([
|
||||
"api",
|
||||
`repos/${repository}/dispatches`,
|
||||
"--method",
|
||||
"POST",
|
||||
"--field",
|
||||
"event_type=missing-model",
|
||||
"--field",
|
||||
`client_payload[provider]=${providerId}`,
|
||||
"--field",
|
||||
`client_payload[issue_number]=${issueNumber}`,
|
||||
]);
|
||||
if (result.code !== 0) {
|
||||
throw new Error(`issue fixer dispatch failed: ${result.stderr || result.stdout || `exit ${result.code}`}`);
|
||||
}
|
||||
}
|
||||
|
||||
async function runGh(args: string[]) {
|
||||
const proc = Bun.spawn(["gh", ...args], {
|
||||
stdout: "pipe",
|
||||
stderr: "pipe",
|
||||
env: process.env,
|
||||
});
|
||||
const [stdout, stderr, code] = await Promise.all([
|
||||
new Response(proc.stdout).text(),
|
||||
new Response(proc.stderr).text(),
|
||||
proc.exited,
|
||||
]);
|
||||
return { code, stdout, stderr };
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { SyncProvider } from "../index.js";
|
||||
import { buildOpenRouterModel, type OpenRouterModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api.ambient.xyz/v1/models";
|
||||
|
||||
export const AmbientModel = z.object({
|
||||
id: z.string().min(1),
|
||||
name: z.string().min(1),
|
||||
created: z.number(),
|
||||
hugging_face_id: z.string().nullable().optional(),
|
||||
context_length: z.number(),
|
||||
max_output_length: z.number(),
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
pricing: z.object({
|
||||
prompt: z.string(),
|
||||
completion: z.string(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
}).passthrough(),
|
||||
supported_features: z.array(z.string()).default([]),
|
||||
supported_sampling_parameters: z.array(z.string()).default([]),
|
||||
openrouter: z.object({ slug: z.string() }).nullable().optional(),
|
||||
is_ready: z.boolean().default(false),
|
||||
}).passthrough();
|
||||
|
||||
export const AmbientResponse = z.object({
|
||||
object: z.literal("list"),
|
||||
data: z.array(AmbientModel),
|
||||
}).passthrough();
|
||||
|
||||
export type AmbientModel = z.infer<typeof AmbientModel>;
|
||||
|
||||
function toOpenRouterShape(model: AmbientModel): OpenRouterModel {
|
||||
return {
|
||||
id: model.openrouter?.slug ?? model.id,
|
||||
name: model.name,
|
||||
created: model.created,
|
||||
hugging_face_id: model.hugging_face_id ?? null,
|
||||
knowledge_cutoff: null,
|
||||
context_length: model.context_length,
|
||||
architecture: {
|
||||
input_modalities: model.input_modalities,
|
||||
output_modalities: model.output_modalities,
|
||||
},
|
||||
pricing: {
|
||||
prompt: model.pricing.prompt,
|
||||
completion: model.pricing.completion,
|
||||
input_cache_read: model.pricing.input_cache_read,
|
||||
input_cache_write: model.pricing.input_cache_write,
|
||||
},
|
||||
top_provider: {
|
||||
context_length: model.context_length,
|
||||
max_completion_tokens: model.max_output_length,
|
||||
},
|
||||
supported_parameters: [...model.supported_features, ...model.supported_sampling_parameters],
|
||||
};
|
||||
}
|
||||
|
||||
export const ambient = {
|
||||
id: "ambient",
|
||||
name: "Ambient",
|
||||
modelsDir: "providers/ambient/models",
|
||||
deleteMissing: false,
|
||||
sourceID(model) {
|
||||
return model.id;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} Ambient models were skipped because the catalog reports them as not ready (is_ready=false).`,
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
missingNotice(paths) {
|
||||
if (paths.length === 0) return [];
|
||||
return [
|
||||
`${paths.length} local Ambient models were absent from the catalog and were retained for manual lifecycle review.`,
|
||||
`Retained local paths: ${paths.map((item) => `\`${item}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
async fetchModels() {
|
||||
const response = await fetch(API_ENDPOINT);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Ambient request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return AmbientResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
if (!model.is_ready) return undefined;
|
||||
const existing = context.existing(model.id);
|
||||
const built = buildOpenRouterModel(toOpenRouterShape(model), existing);
|
||||
const reasoning = model.supported_features.includes("reasoning");
|
||||
const withOptions = reasoning
|
||||
? { ...built, reasoning_options: existing?.reasoning_options ?? [] }
|
||||
: built;
|
||||
const aliasName = ambientAliasName(model.id);
|
||||
return {
|
||||
id: model.id,
|
||||
model: aliasName === undefined ? withOptions : { ...withOptions, name: aliasName },
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<AmbientModel>;
|
||||
|
||||
function ambientAliasName(id: string): string | undefined {
|
||||
if (!id.startsWith("ambient/")) return undefined;
|
||||
const label = id.slice("ambient/".length)
|
||||
.split(/[/-]/)
|
||||
.map((word) => word.charAt(0).toUpperCase() + word.slice(1))
|
||||
.join(" ");
|
||||
return `Ambient ${label}`;
|
||||
}
|
||||
@@ -2,7 +2,8 @@ import path from "node:path";
|
||||
import { existsSync } from "node:fs";
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api.anthropic.com/v1/models";
|
||||
const PRICING_ENDPOINT = "https://platform.claude.com/docs/en/about-claude/pricing";
|
||||
@@ -310,20 +311,42 @@ export function buildAnthropicModel(
|
||||
const output = model.max_tokens > 0 ? model.max_tokens : existing?.limit?.output;
|
||||
const cost = syncedCost(model, existing);
|
||||
const options = reasoningOptions(model, existing);
|
||||
// Models API has no fast-mode surface; preserve authored experimental/provider config.
|
||||
const experimental = existing?.experimental;
|
||||
const provider = existing?.provider;
|
||||
const status = model.pricing?.deprecated ? "deprecated" as const : existing?.status;
|
||||
const structured_output = model.capabilities.structured_outputs?.supported
|
||||
?? existing?.structured_output;
|
||||
const limit = context !== undefined || output !== undefined || existing?.limit !== undefined
|
||||
? {
|
||||
context: context ?? existing?.limit?.context ?? 0,
|
||||
input: existing?.limit?.input,
|
||||
output: output ?? existing?.limit?.output ?? 0,
|
||||
}
|
||||
: undefined;
|
||||
const modalities = { input, output: ["text" as const] };
|
||||
|
||||
if (baseModel !== undefined) {
|
||||
return {
|
||||
base_model: baseModel,
|
||||
const overrides: Partial<SyncedFullModel> = {
|
||||
name: model.canonical_id === undefined ? undefined : name,
|
||||
attachment: input.length > 1,
|
||||
reasoning,
|
||||
reasoning_options: options,
|
||||
structured_output: model.capabilities.structured_outputs?.supported,
|
||||
status: model.pricing?.deprecated ? "deprecated" : undefined,
|
||||
structured_output,
|
||||
status,
|
||||
interleaved: existing?.interleaved,
|
||||
experimental,
|
||||
provider,
|
||||
cost,
|
||||
limit: context !== undefined && output !== undefined ? { context, output } : undefined,
|
||||
modalities: { input, output: ["text"] },
|
||||
limit,
|
||||
modalities,
|
||||
};
|
||||
return factorBaseModel(
|
||||
baseModel,
|
||||
overrides,
|
||||
limit ?? { context: 0, output: 0 },
|
||||
existing?.base_model_omit,
|
||||
);
|
||||
}
|
||||
|
||||
if (
|
||||
@@ -349,15 +372,15 @@ export function buildAnthropicModel(
|
||||
reasoning_options: options,
|
||||
temperature: existing.temperature,
|
||||
tool_call: existing.tool_call,
|
||||
structured_output: model.capabilities.structured_outputs?.supported ?? existing.structured_output,
|
||||
structured_output,
|
||||
knowledge: existing.knowledge,
|
||||
open_weights: existing.open_weights,
|
||||
status: model.pricing?.deprecated ? "deprecated" : existing.status,
|
||||
status,
|
||||
interleaved: existing.interleaved,
|
||||
experimental: existing.experimental,
|
||||
provider: existing.provider,
|
||||
experimental,
|
||||
provider,
|
||||
cost,
|
||||
limit: { context, input: existing.limit?.input, output },
|
||||
modalities: { input, output: ["text"] },
|
||||
modalities,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -81,6 +81,7 @@ const AUTHOR_BY_VENDOR: Record<string, string> = {
|
||||
xiaomi: "xiaomi",
|
||||
minimax: "minimax",
|
||||
"z-ai": "zhipuai",
|
||||
"x-ai": "xai",
|
||||
tencent: "tencent",
|
||||
};
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@ import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
const MODELS_API = "https://api.digitalocean.com/v2/gen-ai/models?per_page=200";
|
||||
const PRICING_API = "https://www.digitalocean.com/api/static-content/v1/products";
|
||||
const CATALOG_API = "https://api.digitalocean.com/v2/gen-ai/models/catalog?limit=200";
|
||||
|
||||
export const DigitalOceanModel = z.object({
|
||||
id: z.string().min(1),
|
||||
@@ -14,6 +14,7 @@ export const DigitalOceanModel = z.object({
|
||||
lifecycle_status: z.string(),
|
||||
type: z.string().optional(),
|
||||
thinking: z.boolean().optional(),
|
||||
reasoning_efforts: z.array(z.string()).optional(),
|
||||
context_window: z.union([z.number(), z.string()]).optional(),
|
||||
modalities: z.object({
|
||||
input: z.array(z.string()).optional(),
|
||||
@@ -36,75 +37,91 @@ const DigitalOceanModelsResponse = z.object({
|
||||
}).passthrough().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const PricingEntry = z.object({
|
||||
name: z.string(),
|
||||
slug: z.string(),
|
||||
model: z.string(),
|
||||
prompt_tokens: z.string().optional(),
|
||||
price: z.object({ rate: z.number() }),
|
||||
const DigitalOceanCatalogPricing = z.object({
|
||||
input_price_per_million: z.number().optional(),
|
||||
output_price_per_million: z.number().optional(),
|
||||
cache_read_input_price_per_million: z.number().optional(),
|
||||
cache_write_5m_input_price_per_million: z.number().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const DigitalOceanPricingResponse = z.object({
|
||||
gradient: z.object({ models: z.array(PricingEntry) }),
|
||||
const DigitalOceanCatalogModel = z.object({
|
||||
id: z.string().min(1).optional(),
|
||||
model_id: z.string().min(1),
|
||||
name: z.string().min(1),
|
||||
context_window: z.union([z.number(), z.string()]).nullish(),
|
||||
max_output_tokens: z.union([z.number(), z.string()]).nullish(),
|
||||
availability: z.array(z.string()).optional(),
|
||||
modalities: z.object({
|
||||
input: z.array(z.string()).optional(),
|
||||
output: z.array(z.string()).optional(),
|
||||
}).nullish(),
|
||||
pricing: DigitalOceanCatalogPricing.nullish(),
|
||||
pricing_detail: z.object({
|
||||
variants: z.array(z.object({
|
||||
tier: z.string().optional(),
|
||||
mode: z.string().optional(),
|
||||
prices: DigitalOceanCatalogPricing.nullish(),
|
||||
}).passthrough()),
|
||||
}).nullish(),
|
||||
}).passthrough();
|
||||
|
||||
const DigitalOceanCatalogResponse = z.object({
|
||||
data: z.array(DigitalOceanCatalogModel),
|
||||
links: z.object({
|
||||
pages: z.object({
|
||||
next: z.string().nullable().optional(),
|
||||
}).passthrough().optional(),
|
||||
}).passthrough().optional(),
|
||||
meta: z.object({
|
||||
page: z.number().int().positive(),
|
||||
pages: z.number().int().nonnegative(),
|
||||
total: z.number().int().nonnegative(),
|
||||
}).passthrough().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const DigitalOceanCatalogDetailResponse = z.object({
|
||||
data: DigitalOceanCatalogModel,
|
||||
}).passthrough();
|
||||
|
||||
const DigitalOceanResponse = z.object({
|
||||
models: z.array(DigitalOceanModel),
|
||||
pricing: z.array(PricingEntry),
|
||||
catalog: z.array(DigitalOceanCatalogModel),
|
||||
});
|
||||
|
||||
export type DigitalOceanModel = z.infer<typeof DigitalOceanModel>;
|
||||
type PricingEntry = z.infer<typeof PricingEntry>;
|
||||
type DigitalOceanCatalogModel = z.infer<typeof DigitalOceanCatalogModel>;
|
||||
|
||||
interface ModelPricing {
|
||||
input?: number;
|
||||
output?: number;
|
||||
inputOver200k?: number;
|
||||
outputOver200k?: number;
|
||||
cacheRead?: number;
|
||||
cacheWrite?: number;
|
||||
extended?: {
|
||||
context: number;
|
||||
input?: number;
|
||||
output?: number;
|
||||
cacheRead?: number;
|
||||
cacheWrite?: number;
|
||||
};
|
||||
}
|
||||
|
||||
type ReasoningEffort =
|
||||
| null
|
||||
| "none"
|
||||
| "minimal"
|
||||
| "low"
|
||||
| "medium"
|
||||
| "high"
|
||||
| "xhigh"
|
||||
| "max"
|
||||
| "default";
|
||||
|
||||
export interface DigitalOceanSourceModel extends DigitalOceanModel {
|
||||
max_output_tokens?: string | number | null;
|
||||
availability?: string[];
|
||||
pricing?: ModelPricing;
|
||||
}
|
||||
|
||||
const PRICING_NAME_OVERRIDES: Record<string, string> = {
|
||||
"claude sonnet 4.6": "anthropic-claude-4.6-sonnet",
|
||||
"claude sonnet 4.5": "anthropic-claude-4.5-sonnet",
|
||||
"claude sonnet 4": "anthropic-claude-sonnet-4",
|
||||
"claude haiku 4.5": "anthropic-claude-haiku-4.5",
|
||||
"claude opus 4.6": "anthropic-claude-opus-4.6",
|
||||
"claude opus 4.5": "anthropic-claude-opus-4.5",
|
||||
"claude opus 4.1": "anthropic-claude-4.1-opus",
|
||||
"claude opus 4": "anthropic-claude-opus-4",
|
||||
"gpt-5.4": "openai-gpt-5.4",
|
||||
"gpt-5.4 mini": "openai-gpt-5.4-mini",
|
||||
"gpt-5.4 nano": "openai-gpt-5.4-nano",
|
||||
"gpt-5.4 pro": "openai-gpt-5.4-pro",
|
||||
"gpt-5.3-codex": "openai-gpt-5.3-codex",
|
||||
"gpt-5.2": "openai-gpt-5.2",
|
||||
"gpt-5.2 pro": "openai-gpt-5.2-pro",
|
||||
"gpt-5.1-codex-max": "openai-gpt-5.1-codex-max",
|
||||
"gpt-5": "openai-gpt-5",
|
||||
"gpt-5 mini": "openai-gpt-5-mini",
|
||||
"gpt-5 nano": "openai-gpt-5-nano",
|
||||
"gpt-4.1": "openai-gpt-4.1",
|
||||
"gpt image 1": "openai-gpt-image-1",
|
||||
"gpt image 1.5": "openai-gpt-image-1.5",
|
||||
"gpt-oss-120b": "openai-gpt-oss-120b",
|
||||
"gpt-oss-20b": "openai-gpt-oss-20b",
|
||||
"gpt-4o": "openai-gpt-4o",
|
||||
"gpt-4o mini": "openai-gpt-4o-mini",
|
||||
o1: "openai-o1",
|
||||
"o3-mini": "openai-o3-mini",
|
||||
"deepseek r1 distill llama 70b": "deepseek-r1-distill-llama-70b",
|
||||
"llama 3.3 70b": "llama3.3-70b-instruct",
|
||||
"qwen3-32b": "alibaba-qwen3-32b",
|
||||
"minimax m2.5": "minimax-m2.5",
|
||||
"kimi k2.5": "kimi-k2.5",
|
||||
"nvidia nemotron 3 super 120b": "nvidia-nemotron-3-super-120b",
|
||||
"glm 5": "glm-5",
|
||||
};
|
||||
|
||||
export const digitalocean = {
|
||||
id: "digitalocean",
|
||||
name: "DigitalOcean",
|
||||
@@ -140,7 +157,7 @@ export const digitalocean = {
|
||||
translateModel(model, context) {
|
||||
const existing = context.existing(model.id);
|
||||
const contextWindow = number(model.context_window);
|
||||
const outputLimit = model.settings?.find((setting) => setting.name === "max_tokens")?.max;
|
||||
const outputLimit = number(model.max_output_tokens ?? undefined);
|
||||
if (model.pricing?.input === undefined || model.pricing.output === undefined) return undefined;
|
||||
if (
|
||||
existing === undefined
|
||||
@@ -162,19 +179,11 @@ export const digitalocean = {
|
||||
} satisfies SyncProvider<DigitalOceanSourceModel>;
|
||||
|
||||
export async function fetchDigitalOceanModels(key: string, fetcher: typeof fetch = fetch) {
|
||||
const [models, pricingResponse] = await Promise.all([
|
||||
const [models, catalog] = await Promise.all([
|
||||
fetchAllDigitalOceanModels(key, fetcher),
|
||||
fetcher(PRICING_API, {
|
||||
headers: { "User-Agent": "models.dev/digitalocean-sync" },
|
||||
}),
|
||||
fetchAllDigitalOceanCatalog(fetcher),
|
||||
]);
|
||||
|
||||
if (!pricingResponse.ok) {
|
||||
throw new Error(`DigitalOcean pricing request failed: ${pricingResponse.status} ${pricingResponse.statusText}`);
|
||||
}
|
||||
|
||||
const pricing = DigitalOceanPricingResponse.parse(await pricingResponse.json()).gradient.models;
|
||||
return { models, pricing };
|
||||
return { models, catalog };
|
||||
}
|
||||
|
||||
async function fetchAllDigitalOceanModels(key: string, fetcher: typeof fetch) {
|
||||
@@ -201,56 +210,124 @@ async function fetchAllDigitalOceanModels(key: string, fetcher: typeof fetch) {
|
||||
return models;
|
||||
}
|
||||
|
||||
async function fetchAllDigitalOceanCatalog(fetcher: typeof fetch) {
|
||||
const catalog: DigitalOceanCatalogModel[] = [];
|
||||
const visited = new Set<string>();
|
||||
let url: string | undefined = CATALOG_API;
|
||||
|
||||
while (url !== undefined) {
|
||||
if (visited.has(url)) throw new Error(`DigitalOcean catalog pagination repeated URL: ${url}`);
|
||||
visited.add(url);
|
||||
|
||||
const response = await fetcher(url, {
|
||||
headers: { "Content-Type": "application/json", "User-Agent": "models.dev/digitalocean-sync" },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`DigitalOcean catalog request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const page = DigitalOceanCatalogResponse.parse(await response.json());
|
||||
catalog.push(...page.data);
|
||||
const next = page.links?.pages?.next;
|
||||
if (next) {
|
||||
url = new URL(next, url).toString();
|
||||
} else if (page.meta !== undefined && page.meta.page < page.meta.pages) {
|
||||
const nextPage = new URL(url);
|
||||
nextPage.searchParams.set("page", String(page.meta.page + 1));
|
||||
url = nextPage.toString();
|
||||
} else {
|
||||
url = undefined;
|
||||
}
|
||||
}
|
||||
|
||||
return Promise.all(catalog.map(async (model) => {
|
||||
if (model.id === undefined || model.availability?.includes("serverless") !== true) return model;
|
||||
const response = await fetcher(`https://api.digitalocean.com/v2/gen-ai/models/catalog/${model.id}`, {
|
||||
headers: { "Content-Type": "application/json", "User-Agent": "models.dev/digitalocean-sync" },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`DigitalOcean catalog detail request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
const detail = DigitalOceanCatalogDetailResponse.parse(await response.json()).data;
|
||||
return {
|
||||
...model,
|
||||
modalities: detail.modalities ?? model.modalities,
|
||||
pricing_detail: detail.pricing_detail ?? model.pricing_detail,
|
||||
};
|
||||
}));
|
||||
}
|
||||
|
||||
export function parseDigitalOceanModels(raw: unknown): DigitalOceanSourceModel[] {
|
||||
const response = DigitalOceanResponse.parse(raw);
|
||||
const pricing = buildPricingMap(response.pricing, response.models);
|
||||
const catalog = new Map(response.catalog.map((model) => [model.model_id, model]));
|
||||
return response.models
|
||||
.filter(isManagedTextModel)
|
||||
.map((model) => ({ ...model, pricing: pricing.get(model.id) }));
|
||||
.map((model) => mergeCatalogModel(model, catalog.get(model.id)))
|
||||
.filter(isManagedTextModel);
|
||||
}
|
||||
|
||||
function isManagedTextModel(model: DigitalOceanModel) {
|
||||
function mergeCatalogModel(
|
||||
model: DigitalOceanModel,
|
||||
catalog: DigitalOceanCatalogModel | undefined,
|
||||
): DigitalOceanSourceModel {
|
||||
return {
|
||||
...model,
|
||||
name: catalog?.name ?? model.name,
|
||||
context_window: catalog?.context_window ?? model.context_window,
|
||||
max_output_tokens: catalog?.max_output_tokens,
|
||||
modalities: catalog?.modalities ?? model.modalities,
|
||||
availability: catalog?.availability,
|
||||
pricing: catalogPricing(catalog),
|
||||
};
|
||||
}
|
||||
|
||||
function isManagedTextModel(model: DigitalOceanSourceModel) {
|
||||
const output = normalizeModalities(model.modalities?.output ?? [], []);
|
||||
return output.includes("text") && model.type !== "embedding" && model.type !== "reranking";
|
||||
return model.availability?.includes("serverless") === true
|
||||
&& output.includes("text")
|
||||
&& model.type !== "embedding"
|
||||
&& model.type !== "reranking";
|
||||
}
|
||||
|
||||
function pricingName(value: string) {
|
||||
return value
|
||||
.replace(/\s+(input|output)\s+tokens$/i, "")
|
||||
.replace(/\s*\(public preview\)\s*/i, " ")
|
||||
.trim()
|
||||
.toLowerCase();
|
||||
function catalogPricing(model: DigitalOceanCatalogModel | undefined): ModelPricing | undefined {
|
||||
if (model?.pricing == null) return undefined;
|
||||
const standard = model.pricing_detail?.variants.find((variant) =>
|
||||
variant.mode === "MODEL_BILLING_MODE_INTERACTIVE"
|
||||
&& variant.tier === "MODEL_PRICING_TIER_STANDARD"
|
||||
)?.prices;
|
||||
const extended = model.pricing_detail?.variants.find((variant) =>
|
||||
variant.mode === "MODEL_BILLING_MODE_INTERACTIVE"
|
||||
&& variant.tier?.startsWith("MODEL_PRICING_TIER_EXTENDED_") === true
|
||||
);
|
||||
const extendedContext = pricingTierContext(extended?.tier);
|
||||
return {
|
||||
input: perMillion(model.pricing.input_price_per_million),
|
||||
output: perMillion(model.pricing.output_price_per_million),
|
||||
cacheRead: perMillion(model.pricing.cache_read_input_price_per_million),
|
||||
cacheWrite: perMillion(standard?.cache_write_5m_input_price_per_million),
|
||||
extended: extendedContext === undefined || extended?.prices == null
|
||||
? undefined
|
||||
: {
|
||||
context: extendedContext,
|
||||
input: perMillion(extended.prices.input_price_per_million),
|
||||
output: perMillion(extended.prices.output_price_per_million),
|
||||
cacheRead: perMillion(extended.prices.cache_read_input_price_per_million),
|
||||
cacheWrite: perMillion(extended.prices.cache_write_5m_input_price_per_million),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function normalizedName(value: string) {
|
||||
return value.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
|
||||
function pricingTierContext(tier: string | undefined) {
|
||||
// Tier names describe capacity; Anthropic's 1M surcharge starts above 200K.
|
||||
if (tier === "MODEL_PRICING_TIER_EXTENDED_1M") return 200_000;
|
||||
if (tier === "MODEL_PRICING_TIER_EXTENDED_272K") return 272_000;
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function buildPricingMap(entries: PricingEntry[], models: DigitalOceanModel[]) {
|
||||
const names = new Map<string, string[]>();
|
||||
for (const model of models) {
|
||||
const key = normalizedName(model.name);
|
||||
names.set(key, [...names.get(key) ?? [], model.id]);
|
||||
}
|
||||
|
||||
const result = new Map<string, ModelPricing>();
|
||||
for (const entry of entries) {
|
||||
const name = pricingName(entry.name);
|
||||
const matches = names.get(normalizedName(name)) ?? [];
|
||||
const id = PRICING_NAME_OVERRIDES[name] ?? (matches.length === 1 ? matches[0] : undefined);
|
||||
if (id === undefined) continue;
|
||||
|
||||
const price = Math.round(entry.price.rate * 10_000) / 10_000;
|
||||
const current = result.get(id) ?? {};
|
||||
const input = /\sinput\s+tokens$/i.test(entry.name);
|
||||
const over200k = entry.prompt_tokens === ">200k";
|
||||
if (input && over200k) current.inputOver200k = price;
|
||||
else if (!input && over200k) current.outputOver200k = price;
|
||||
else if (input) current.input = price;
|
||||
else current.output = price;
|
||||
result.set(id, current);
|
||||
}
|
||||
return result;
|
||||
function perMillion(value: number | undefined) {
|
||||
if (value === undefined) return undefined;
|
||||
// The live catalog currently returns per-token rates despite the field names.
|
||||
const normalized = value < 0.001 ? value * 1_000_000 : value;
|
||||
return Math.round(normalized * 10_000) / 10_000;
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
@@ -279,6 +356,40 @@ function inferFamily(id: string, name: string) {
|
||||
.find((family) => target.includes(family.toLowerCase()));
|
||||
}
|
||||
|
||||
function reasoningOptionsFor(
|
||||
model: DigitalOceanSourceModel,
|
||||
existing: ExistingModel | undefined,
|
||||
): ExistingModel["reasoning_options"] {
|
||||
if (model.reasoning_efforts === undefined) return existing?.reasoning_options;
|
||||
const values = model.reasoning_efforts
|
||||
.map((value) => value === "null" ? null : value)
|
||||
.filter(isReasoningEffort);
|
||||
const preserved = existing?.reasoning_options?.filter((option) => option.type !== "effort") ?? [];
|
||||
return values.length > 0 ? [...preserved, { type: "effort", values }] : preserved;
|
||||
}
|
||||
|
||||
function isReasoningEffort(value: string | null): value is ReasoningEffort {
|
||||
return value === null
|
||||
|| value === "none"
|
||||
|| value === "minimal"
|
||||
|| value === "low"
|
||||
|| value === "medium"
|
||||
|| value === "high"
|
||||
|| value === "xhigh"
|
||||
|| value === "max"
|
||||
|| value === "default";
|
||||
}
|
||||
|
||||
function status(
|
||||
lifecycleStatus: string,
|
||||
existing: ExistingModel["status"],
|
||||
): ExistingModel["status"] {
|
||||
const lifecycle = lifecycleStatus.toLowerCase().replaceAll("_", "-");
|
||||
if (lifecycle === "deprecated" || lifecycle === "end-of-life") return "deprecated";
|
||||
if (lifecycle === "public-preview") return "beta";
|
||||
return existing === "deprecated" || existing === "beta" ? undefined : existing;
|
||||
}
|
||||
|
||||
function cost(model: DigitalOceanSourceModel, existing: ExistingModel | undefined) {
|
||||
const input = model.pricing?.input ?? existing?.cost?.input;
|
||||
const output = model.pricing?.output ?? existing?.cost?.output;
|
||||
@@ -288,18 +399,18 @@ function cost(model: DigitalOceanSourceModel, existing: ExistingModel | undefine
|
||||
const longContext = existingTiers.find((tier) =>
|
||||
(tier.tier.type === undefined || tier.tier.type === "context") && tier.tier.size >= 200_000
|
||||
);
|
||||
const hasLongContextPricing = model.pricing?.inputOver200k !== undefined
|
||||
&& model.pricing.outputOver200k !== undefined;
|
||||
const extended = model.pricing?.extended;
|
||||
const hasLongContextPricing = extended?.input !== undefined && extended.output !== undefined;
|
||||
const tiers = hasLongContextPricing
|
||||
? [
|
||||
...existingTiers.filter((tier) => tier !== longContext),
|
||||
{
|
||||
tier: { type: "context" as const, size: longContext?.tier.size ?? 200_000 },
|
||||
input: model.pricing!.inputOver200k!,
|
||||
output: model.pricing!.outputOver200k!,
|
||||
tier: { type: "context" as const, size: extended.context },
|
||||
input: extended.input!,
|
||||
output: extended.output!,
|
||||
reasoning: longContext?.reasoning,
|
||||
cache_read: longContext?.cache_read,
|
||||
cache_write: longContext?.cache_write,
|
||||
cache_read: extended.cacheRead ?? longContext?.cache_read,
|
||||
cache_write: extended.cacheWrite ?? longContext?.cache_write,
|
||||
},
|
||||
]
|
||||
: existingTiers;
|
||||
@@ -308,8 +419,8 @@ function cost(model: DigitalOceanSourceModel, existing: ExistingModel | undefine
|
||||
input,
|
||||
output,
|
||||
reasoning: existing?.cost?.reasoning,
|
||||
cache_read: existing?.cost?.cache_read,
|
||||
cache_write: existing?.cost?.cache_write,
|
||||
cache_read: model.pricing?.cacheRead ?? existing?.cost?.cache_read,
|
||||
cache_write: model.pricing?.cacheWrite ?? existing?.cost?.cache_write,
|
||||
input_audio: existing?.cost?.input_audio,
|
||||
output_audio: existing?.cost?.output_audio,
|
||||
tiers: tiers.length > 0 ? tiers : undefined,
|
||||
@@ -330,14 +441,19 @@ export function buildDigitalOceanModel(
|
||||
existing?.modalities?.output ?? ["text"],
|
||||
);
|
||||
const context = number(model.context_window) ?? existing?.limit?.context ?? 0;
|
||||
const maxTokens = model.settings?.find((setting) => setting.name === "max_tokens")?.max;
|
||||
const maxTokens = number(model.max_output_tokens ?? undefined);
|
||||
const limit = {
|
||||
context,
|
||||
input: existing?.limit?.input,
|
||||
output: maxTokens ?? existing?.limit?.output ?? 0,
|
||||
};
|
||||
const textOutput = output.includes("text") && !output.includes("image") && !output.includes("video");
|
||||
const reasoning = existing?.reasoning ?? (textOutput && (model.thinking ?? false));
|
||||
const remoteReasoning = textOutput
|
||||
&& ((model.thinking ?? false) || (model.reasoning_efforts?.length ?? 0) > 0);
|
||||
const providerReasoning = remoteReasoning ? true : existing?.reasoning;
|
||||
const reasoning = providerReasoning ?? false;
|
||||
const reasoningOptions = reasoning ? reasoningOptionsFor(model, existing) : undefined;
|
||||
const modelStatus = status(model.lifecycle_status, existing?.status);
|
||||
const releaseDate = existing?.release_date ?? model.created_at?.slice(0, 10) ?? new Date().toISOString().slice(0, 10);
|
||||
const values: Partial<SyncedFullModel> = {
|
||||
name: model.name,
|
||||
@@ -357,15 +473,13 @@ export function buildDigitalOceanModel(
|
||||
last_updated: existing?.last_updated ?? releaseDate,
|
||||
attachment: existing?.attachment ?? input.some((value) => value !== "text"),
|
||||
reasoning,
|
||||
reasoning_options: existing?.reasoning_options,
|
||||
reasoning_options: reasoningOptions,
|
||||
temperature: existing?.temperature ?? true,
|
||||
tool_call: existing?.tool_call ?? textOutput,
|
||||
structured_output: existing?.structured_output,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: existing?.open_weights ?? false,
|
||||
status: model.lifecycle_status === "end_of_life"
|
||||
? "deprecated"
|
||||
: existing?.status === "deprecated" ? undefined : existing?.status,
|
||||
status: modelStatus,
|
||||
interleaved: existing?.interleaved,
|
||||
cost: cost(model, existing),
|
||||
limit,
|
||||
@@ -379,14 +493,12 @@ export function buildDigitalOceanModel(
|
||||
name: model.name,
|
||||
description: existing?.description,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning: model.thinking ?? existing?.reasoning,
|
||||
reasoning_options: existing?.reasoning_options,
|
||||
reasoning: providerReasoning,
|
||||
reasoning_options: reasoningOptions,
|
||||
temperature: existing?.temperature,
|
||||
tool_call: existing?.tool_call,
|
||||
structured_output: existing?.structured_output,
|
||||
status: model.lifecycle_status === "end_of_life"
|
||||
? "deprecated"
|
||||
: existing?.status === "deprecated" ? undefined : existing?.status,
|
||||
status: modelStatus,
|
||||
interleaved: existing?.interleaved,
|
||||
cost: cost(model, existing),
|
||||
limit,
|
||||
|
||||
@@ -0,0 +1,367 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
// EmpirioLabs exposes a public, unauthenticated OpenAI-compatible model
|
||||
// catalog, so no API key is needed or used for this sync.
|
||||
const API_ENDPOINT = "https://api.empiriolabs.ai/v1/models";
|
||||
|
||||
const CANONICAL_BASE_MODELS: Record<string, string> = {
|
||||
"fugu-ultra": "sakana/fugu-ultra",
|
||||
"gemma-4-26b-a4b": "google/gemma-4-26b-a4b-it",
|
||||
"gemma-4-e4b": "google/gemma-4-E4B-it",
|
||||
"mistral-medium-3": "mistral/mistral-medium-2505",
|
||||
"mistral-small-4": "mistral/mistral-small-2603",
|
||||
"muse-spark-1-1": "meta/muse-spark-1.1",
|
||||
"qwen3-5-9b": "alibaba/qwen3.5-9b",
|
||||
"qwen3-7-max": "alibaba/qwen3.7-max",
|
||||
"qwen3-7-plus": "alibaba/qwen3.7-plus",
|
||||
"step-3-5-flash": "stepfun/step-3.5-flash",
|
||||
"step-3-5-flash-2603": "stepfun/step-3.5-flash-2603",
|
||||
"step-3-7-flash": "stepfun/step-3.7-flash",
|
||||
};
|
||||
|
||||
const EmpiriolabsParameter = z
|
||||
.object({
|
||||
name: z.string(),
|
||||
type: z.string().optional(),
|
||||
options: z.array(z.string()).optional(),
|
||||
min: z.number().optional(),
|
||||
max: z.number().optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const EmpiriolabsPricingTier = z
|
||||
.object({
|
||||
prompt: z.string().optional(),
|
||||
completion: z.string().optional(),
|
||||
input_cache_read: z.string().optional(),
|
||||
min_context: z.number().nullable().optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
// Pricing is returned either as a single tier object or as an array of tier
|
||||
// objects (tiered/context-priced models). Accept both shapes.
|
||||
const EmpiriolabsPricing = z.union([
|
||||
z.array(EmpiriolabsPricingTier),
|
||||
EmpiriolabsPricingTier,
|
||||
]);
|
||||
|
||||
const EmpiriolabsModel = z
|
||||
.object({
|
||||
id: z.string(),
|
||||
display_name: z.string().optional(),
|
||||
name: z.string().optional(),
|
||||
description: z.string().optional(),
|
||||
category: z.string().optional(),
|
||||
context_length: z.number().nullable().optional(),
|
||||
context_window: z.number().nullable().optional(),
|
||||
max_output_tokens: z.number().nullable().optional(),
|
||||
model_released_at: z.string().nullable().optional(),
|
||||
pricing: EmpiriolabsPricing.optional(),
|
||||
capabilities: z.record(z.unknown()).optional(),
|
||||
features: z.array(z.string()).optional(),
|
||||
structured_output: z.string().nullable().optional(),
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
supported_parameters: z.array(EmpiriolabsParameter).optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const EmpiriolabsResponse = z
|
||||
.object({
|
||||
data: z.array(EmpiriolabsModel),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
export type EmpiriolabsModel = z.infer<typeof EmpiriolabsModel>;
|
||||
|
||||
export const empiriolabs = {
|
||||
id: "empiriolabs",
|
||||
name: "EmpirioLabs AI",
|
||||
modelsDir: "providers/empiriolabs/models",
|
||||
sourceID(model) {
|
||||
return model.id;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} EmpirioLabs AI models returned by the API were not created because they could not be mapped exactly to models.dev canonical metadata. `
|
||||
+ "Existing models and canonical matches are still updated from API-authoritative fields.",
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
async fetchModels() {
|
||||
const response = await fetch(API_ENDPOINT);
|
||||
if (!response.ok) {
|
||||
throw new Error(`EmpirioLabs request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
// Text chat models only. Skip non-text categories (image, video, audio,
|
||||
// 3D, research, tools) and regional/capability variant lanes (id has ":").
|
||||
return EmpiriolabsResponse.parse(raw).data.filter(
|
||||
(model) => (model.category ?? "").toLowerCase() === "text" && !model.id.includes(":"),
|
||||
);
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const existing = context.existing(model.id);
|
||||
const baseModel = existing?.base_model ?? resolveEmpiriolabsBaseModel(model.id);
|
||||
if (existing === undefined && baseModel === undefined) return undefined;
|
||||
const built = buildEmpiriolabsModel(model, existing, baseModel);
|
||||
// A model with no resolvable context window cannot produce a valid TOML
|
||||
// (limit.context is required), so skip it rather than fail the whole sync.
|
||||
if (built === undefined) return undefined;
|
||||
return {
|
||||
id: model.id,
|
||||
model: built,
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<EmpiriolabsModel>;
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
type EffortValue =
|
||||
| "none"
|
||||
| "minimal"
|
||||
| "low"
|
||||
| "medium"
|
||||
| "high"
|
||||
| "xhigh"
|
||||
| "max"
|
||||
| "default";
|
||||
|
||||
const EFFORT_VALUES: EffortValue[] = [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max",
|
||||
"default",
|
||||
];
|
||||
|
||||
function price(value: string | undefined) {
|
||||
if (value === undefined) return undefined;
|
||||
const number = Number(value);
|
||||
// Per-token string converted to a per-1M-token number.
|
||||
return Number.isFinite(number) && number >= 0
|
||||
? Math.round(number * 1_000_000_000_000) / 1_000_000
|
||||
: undefined;
|
||||
}
|
||||
|
||||
function nonZeroPrice(value: string | undefined) {
|
||||
const result = price(value);
|
||||
return result !== undefined && result > 0 ? result : undefined;
|
||||
}
|
||||
|
||||
type TierCost = { input: number; output: number; cache_read?: number };
|
||||
|
||||
function tierCost(tier: z.infer<typeof EmpiriolabsPricingTier> | undefined): TierCost | undefined {
|
||||
const input = price(tier?.prompt);
|
||||
const output = price(tier?.completion);
|
||||
if (input === undefined || output === undefined) return undefined;
|
||||
const cacheRead = nonZeroPrice(tier?.input_cache_read);
|
||||
return { input, output, cache_read: cacheRead };
|
||||
}
|
||||
|
||||
function modalities(values: string[] | undefined, fallback: Modality[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = (values ?? [])
|
||||
.map((value) => value.toLowerCase())
|
||||
.map((value) => (value === "file" ? "pdf" : value))
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
|
||||
function reasoningOptions(model: EmpiriolabsModel): SyncedModel["reasoning_options"] {
|
||||
const params = model.supported_parameters ?? [];
|
||||
const options: NonNullable<SyncedModel["reasoning_options"]> = [];
|
||||
if (params.some((parameter) => parameter.name === "enable_thinking")) {
|
||||
options.push({ type: "toggle" });
|
||||
}
|
||||
|
||||
const effort = params.find((parameter) => parameter.name === "reasoning_effort");
|
||||
if (effort?.options?.length) {
|
||||
const values = effort.options.filter((value): value is EffortValue =>
|
||||
(EFFORT_VALUES as string[]).includes(value),
|
||||
);
|
||||
if (values.length > 0) options.push({ type: "effort", values });
|
||||
}
|
||||
|
||||
const budget = params.find((parameter) => parameter.name === "thinking_budget");
|
||||
if (budget !== undefined) {
|
||||
const option: { type: "budget_tokens"; min?: number; max?: number } = { type: "budget_tokens" };
|
||||
if (budget.min !== undefined) option.min = budget.min;
|
||||
if (budget.max !== undefined) option.max = budget.max;
|
||||
options.push(option);
|
||||
}
|
||||
return options;
|
||||
}
|
||||
|
||||
function parameterOutputLimit(model: EmpiriolabsModel) {
|
||||
const parameter = (model.supported_parameters ?? []).find(
|
||||
(item) => item.name === "max_tokens" || item.name === "max_completion_tokens",
|
||||
);
|
||||
return parameter?.max !== undefined && parameter.max > 0 ? parameter.max : undefined;
|
||||
}
|
||||
|
||||
export function resolveEmpiriolabsBaseModel(id: string) {
|
||||
const explicit = CANONICAL_BASE_MODELS[id];
|
||||
if (explicit !== undefined) return explicit;
|
||||
return canonicalCandidates(id)
|
||||
.map((candidate) => resolveCanonicalBaseModel(candidate))
|
||||
.find((candidate) => candidate !== undefined);
|
||||
}
|
||||
|
||||
function canonicalCandidates(id: string) {
|
||||
const candidates: string[] = [];
|
||||
|
||||
if (id.startsWith("deepseek-")) {
|
||||
candidates.push(`deepseek/${id}`);
|
||||
candidates.push(`deepseek/${id.replace(/^deepseek-v(\d+)-(\d+)/, "deepseek-v$1.$2")}`);
|
||||
}
|
||||
|
||||
if (id.startsWith("glm-")) {
|
||||
const normalized = id
|
||||
.replace(/^glm-(\d+)-(\d+)/, "glm-$1.$2")
|
||||
.replace(/^glm-(\d+)-(\d+)v/, "glm-$1.$2v");
|
||||
candidates.push(`z-ai/${id}`);
|
||||
candidates.push(`z-ai/${normalized}`);
|
||||
}
|
||||
|
||||
if (id.startsWith("kimi-")) {
|
||||
const normalized = id.replace(/^(kimi-k\d+)-(\d+)/, "$1.$2");
|
||||
candidates.push(`moonshotai/${id}`);
|
||||
candidates.push(`moonshotai/${normalized}`);
|
||||
}
|
||||
|
||||
if (id.startsWith("minimax-")) {
|
||||
const normalized = id.replace(/^minimax-m(\d+)-(\d+)/, "minimax-m$1.$2");
|
||||
candidates.push(`minimax/${id}`);
|
||||
candidates.push(`minimax/${normalized}`);
|
||||
}
|
||||
|
||||
if (id.startsWith("mimo-")) {
|
||||
const normalized = id.replace(/^mimo-v(\d+)-(\d+)/, "mimo-v$1.$2");
|
||||
candidates.push(`xiaomi/${id}`);
|
||||
candidates.push(`xiaomi/${normalized}`);
|
||||
}
|
||||
|
||||
if (id.startsWith("qwen")) {
|
||||
const normalized = id.replace(/^(qwen\d+)-(\d+)/, "$1.$2");
|
||||
candidates.push(`qwen/${id}`);
|
||||
candidates.push(`qwen/${normalized}`);
|
||||
}
|
||||
|
||||
return [...new Set(candidates)];
|
||||
}
|
||||
|
||||
export function buildEmpiriolabsModel(
|
||||
model: EmpiriolabsModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel = existing?.base_model ?? resolveEmpiriolabsBaseModel(model.id),
|
||||
): SyncedModel | undefined {
|
||||
const features = new Set(model.features ?? []);
|
||||
const capabilities = (model.capabilities ?? {}) as Record<string, unknown>;
|
||||
const input = modalities(model.input_modalities, ["text"]);
|
||||
const output = modalities(model.output_modalities, ["text"]);
|
||||
const attachment = input.some((value) => value !== "text");
|
||||
const reasoning =
|
||||
capabilities.reasoning === true || features.has("reasoning") || existing?.reasoning === true;
|
||||
const toolCall =
|
||||
features.has("function_calling") || features.has("tools") || existing?.tool_call === true;
|
||||
const structuredOutput = features.has("structured_output") || existing?.structured_output === true;
|
||||
const temperature =
|
||||
(model.supported_parameters ?? []).some((parameter) => parameter.name === "temperature")
|
||||
|| existing?.temperature === true;
|
||||
|
||||
const pricingTiers = model.pricing === undefined
|
||||
? []
|
||||
: Array.isArray(model.pricing)
|
||||
? [...model.pricing].sort((a, b) => (a.min_context ?? 0) - (b.min_context ?? 0))
|
||||
: [model.pricing];
|
||||
const baseCost = tierCost(pricingTiers[0]);
|
||||
const contextTiers = pricingTiers
|
||||
.slice(1)
|
||||
.map((tier) => {
|
||||
const tierPricing = tierCost(tier);
|
||||
return tierPricing === undefined || tier.min_context === undefined || tier.min_context === null
|
||||
? undefined
|
||||
: { tier: { type: "context" as const, size: tier.min_context }, ...tierPricing };
|
||||
})
|
||||
.filter((tier): tier is NonNullable<typeof tier> => tier !== undefined);
|
||||
const cost = baseCost !== undefined
|
||||
? {
|
||||
...baseCost,
|
||||
reasoning: existing?.cost?.reasoning,
|
||||
cache_write: existing?.cost?.cache_write,
|
||||
tiers: contextTiers.length > 0 ? contextTiers : undefined,
|
||||
}
|
||||
: existing?.cost;
|
||||
|
||||
const context =
|
||||
model.context_length ?? model.context_window ?? existing?.limit?.context;
|
||||
// No usable context window: cannot build a valid model TOML, so skip.
|
||||
if (context === undefined || context === null) return undefined;
|
||||
|
||||
const releaseDate = baseModel === undefined
|
||||
? model.model_released_at ?? existing?.release_date
|
||||
: undefined;
|
||||
const lastUpdated = baseModel === undefined
|
||||
? model.model_released_at ?? existing?.last_updated ?? releaseDate
|
||||
: existing?.last_updated ?? releaseDate;
|
||||
const outputTokens = model.max_output_tokens
|
||||
?? parameterOutputLimit(model)
|
||||
?? existing?.limit?.output
|
||||
?? context;
|
||||
const limit = {
|
||||
context,
|
||||
input: existing?.limit?.input,
|
||||
output: outputTokens,
|
||||
};
|
||||
const values: Partial<SyncedFullModel> = {
|
||||
name: model.display_name ?? model.name ?? model.id,
|
||||
description: baseModel === undefined ? existing?.description ?? model.description : existing?.description,
|
||||
family: existing?.family,
|
||||
release_date: releaseDate,
|
||||
last_updated: lastUpdated,
|
||||
attachment,
|
||||
reasoning,
|
||||
reasoning_options: reasoning ? reasoningOptions(model) : undefined,
|
||||
temperature: temperature || undefined,
|
||||
tool_call: toolCall,
|
||||
structured_output:
|
||||
(model.structured_output !== undefined && model.structured_output !== null)
|
||||
|| structuredOutput
|
||||
|| undefined,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: existing?.open_weights,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
};
|
||||
|
||||
if (baseModel !== undefined) {
|
||||
return factorBaseModel(baseModel, values, limit, existing?.base_model_omit);
|
||||
}
|
||||
|
||||
if (existing === undefined) return undefined;
|
||||
const required = z.object({
|
||||
name: z.string(),
|
||||
description: z.string(),
|
||||
release_date: z.string(),
|
||||
last_updated: z.string(),
|
||||
open_weights: z.boolean(),
|
||||
cost: z.object({ input: z.number(), output: z.number() }),
|
||||
}).safeParse(values);
|
||||
if (!required.success) {
|
||||
throw new Error(`EmpirioLabs model ${model.id} has incomplete local metadata required for sync`);
|
||||
}
|
||||
|
||||
return values as SyncedFullModel;
|
||||
}
|
||||
@@ -29,13 +29,31 @@ const GoogleResponse = z.object({
|
||||
|
||||
type GoogleModel = z.infer<typeof GoogleModel>;
|
||||
|
||||
const TrackedModelPrefixes = [
|
||||
"deep-research-",
|
||||
"gemini-",
|
||||
"gemma-",
|
||||
"imagen-",
|
||||
"lyria-",
|
||||
"nano-banana-",
|
||||
"veo-",
|
||||
];
|
||||
|
||||
export function shouldTrackGoogleModel(id: string) {
|
||||
return TrackedModelPrefixes.some((prefix) => id.startsWith(prefix));
|
||||
}
|
||||
|
||||
export const google = {
|
||||
id: "google",
|
||||
name: "Google",
|
||||
modelsDir: "providers/google/models",
|
||||
skipCreates: true,
|
||||
// /v1beta/models has no lifecycle fields and can retain shut-down,
|
||||
// superseded, moving-alias, and EAP model IDs.
|
||||
trackMissingModels: false,
|
||||
sourceID(model) {
|
||||
return model.name.replace(/^models\//, "");
|
||||
const id = model.name.replace(/^models\//, "");
|
||||
return shouldTrackGoogleModel(id) ? id : undefined;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
import { existsSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import { z } from "zod";
|
||||
|
||||
import { describeModel } from "../../describe.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveModelMetadataBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://hyper.charm.land/v1/models";
|
||||
const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models");
|
||||
|
||||
function baseModelExists(modelID: string) {
|
||||
return existsSync(path.join(MODELS_DIR, `${modelID}.toml`));
|
||||
}
|
||||
|
||||
function resolveHyperBaseModel(modelID: string, existingBase: string | undefined) {
|
||||
if (existingBase !== undefined && baseModelExists(existingBase)) return existingBase;
|
||||
const resolved = resolveModelMetadataBaseModel(modelID);
|
||||
return resolved !== undefined && baseModelExists(resolved) ? resolved : undefined;
|
||||
}
|
||||
|
||||
const ReasoningEffort = z.enum([
|
||||
"default",
|
||||
"max",
|
||||
"low",
|
||||
"high",
|
||||
"none",
|
||||
"medium",
|
||||
"minimal",
|
||||
"xhigh",
|
||||
]);
|
||||
|
||||
export const HyperModel = z.object({
|
||||
id: z.string(),
|
||||
created: z.number(),
|
||||
display_name: z.string(),
|
||||
context_window: z.number(),
|
||||
max_output_tokens: z.number(),
|
||||
capabilities: z.object({
|
||||
vision: z.boolean().optional(),
|
||||
}).optional(),
|
||||
reasoning: z.object({
|
||||
effort_levels: z.array(z.object({
|
||||
value: z.string(),
|
||||
display: z.string().optional(),
|
||||
})).optional(),
|
||||
}).optional(),
|
||||
pricing: z.object({
|
||||
input: z.number().optional(),
|
||||
output: z.number().optional(),
|
||||
cache_hit: z.number().optional(),
|
||||
cache_create: z.number().optional(),
|
||||
}).optional(),
|
||||
}).passthrough();
|
||||
|
||||
const HyperResponse = z.object({
|
||||
data: z.array(HyperModel),
|
||||
}).passthrough();
|
||||
|
||||
export type HyperModel = z.infer<typeof HyperModel>;
|
||||
|
||||
export const hyper = {
|
||||
id: "hyper",
|
||||
name: "Charm Hyper",
|
||||
modelsDir: "providers/hyper/models",
|
||||
preserveBaseModels: false,
|
||||
async fetchModels() {
|
||||
const key = process.env.HYPER_API_KEY;
|
||||
const response = await fetch(API_ENDPOINT, key
|
||||
? { headers: { Authorization: `Bearer ${key}` } }
|
||||
: undefined);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Hyper models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return HyperResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const existing = context.existing(model.id);
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildHyperModel(model, existing),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<HyperModel>;
|
||||
|
||||
function dateFromTimestamp(timestamp: number) {
|
||||
return new Date(timestamp * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function reasoningOptions(model: HyperModel) {
|
||||
const effortLevels = model.reasoning?.effort_levels?.map((level) => level.value) ?? [];
|
||||
if (effortLevels.length === 0) return [];
|
||||
const values = effortLevels.filter(isReasoningEffort);
|
||||
if (values.length === 0) return [{ type: "toggle" as const }];
|
||||
return [{ type: "effort" as const, values }];
|
||||
}
|
||||
|
||||
function isReasoningEffort(value: string): value is z.infer<typeof ReasoningEffort> {
|
||||
return ReasoningEffort.safeParse(value).success;
|
||||
}
|
||||
|
||||
function price(value: number) {
|
||||
return Math.round(value * 1_000_000) / 1_000_000;
|
||||
}
|
||||
|
||||
function positivePrice(value: number | undefined) {
|
||||
return value !== undefined && value > 0 ? price(value) : undefined;
|
||||
}
|
||||
|
||||
function buildCost(model: HyperModel, existing: ExistingModel["cost"] | undefined) {
|
||||
const pricing = model.pricing;
|
||||
if (pricing?.input === undefined || pricing.output === undefined) return existing;
|
||||
|
||||
return {
|
||||
input: price(pricing.input),
|
||||
output: price(pricing.output),
|
||||
cache_read: positivePrice(pricing.cache_hit)
|
||||
?? (pricing.cache_hit === undefined ? existing?.cache_read : undefined),
|
||||
cache_write: positivePrice(pricing.cache_create)
|
||||
?? (pricing.cache_create === undefined ? existing?.cache_write : undefined),
|
||||
reasoning: existing?.reasoning,
|
||||
};
|
||||
}
|
||||
|
||||
function hyperModalities(vision: boolean) {
|
||||
const input = vision ? ["text" as const, "image" as const] : ["text" as const];
|
||||
return {
|
||||
input,
|
||||
output: ["text" as const],
|
||||
};
|
||||
}
|
||||
|
||||
export function buildHyperModel(
|
||||
model: HyperModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel = existing?.base_model,
|
||||
today = new Date().toISOString().slice(0, 10),
|
||||
): SyncedModel {
|
||||
const limit = {
|
||||
context: model.context_window,
|
||||
input: existing?.limit?.input,
|
||||
output: model.max_output_tokens,
|
||||
};
|
||||
const modalities = hyperModalities(model.capabilities?.vision ?? false);
|
||||
const reasoning = model.reasoning != null;
|
||||
const releaseDate = existing?.release_date ?? dateFromTimestamp(model.created);
|
||||
const values: Partial<SyncedFullModel> = {
|
||||
attachment: modalities.input.some((value) => value !== "text"),
|
||||
modalities,
|
||||
reasoning,
|
||||
release_date: releaseDate,
|
||||
last_updated: existing?.last_updated ?? today,
|
||||
interleaved: existing?.interleaved,
|
||||
cost: buildCost(model, existing?.cost),
|
||||
limit,
|
||||
};
|
||||
if (reasoning) values.reasoning_options = reasoningOptions(model);
|
||||
|
||||
const resolvedBase = resolveHyperBaseModel(model.id, baseModel);
|
||||
if (resolvedBase !== undefined) {
|
||||
return factorBaseModel(
|
||||
resolvedBase,
|
||||
values,
|
||||
limit,
|
||||
existing?.base_model === resolvedBase ? existing.base_model_omit : undefined,
|
||||
);
|
||||
}
|
||||
|
||||
const name = existing?.name ?? model.display_name;
|
||||
return {
|
||||
name,
|
||||
description: existing?.description ?? describeModel({
|
||||
id: model.id,
|
||||
name,
|
||||
family: existing?.family,
|
||||
reasoning,
|
||||
tool_call: existing?.tool_call ?? true,
|
||||
structured_output: existing?.structured_output,
|
||||
open_weights: existing?.open_weights ?? false,
|
||||
limit,
|
||||
modalities,
|
||||
}),
|
||||
family: existing?.family,
|
||||
...values,
|
||||
temperature: existing?.temperature,
|
||||
tool_call: existing?.tool_call ?? true,
|
||||
structured_output: existing?.structured_output,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: existing?.open_weights ?? false,
|
||||
status: existing?.status,
|
||||
provider: existing?.provider,
|
||||
experimental: existing?.experimental,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,428 @@
|
||||
import { z } from "zod";
|
||||
import { readFileSync, readdirSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
|
||||
import { describeModel } from "../../describe.js";
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api.kilo.ai/api/gateway/models";
|
||||
const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models");
|
||||
const modelMetadataByID = new Map<string, Record<string, unknown>>();
|
||||
const modelMetadataFilesByProvider = new Map<string, Set<string>>();
|
||||
|
||||
|
||||
export const KiloModel = z.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
created: z.number(),
|
||||
description: z.string().optional(),
|
||||
hugging_face_id: z.string().nullable().optional(),
|
||||
knowledge_cutoff: z.string().nullable().optional(),
|
||||
context_length: z.number(),
|
||||
architecture: z.object({
|
||||
modality: z.string().optional(),
|
||||
input_modalities: z.array(z.string()),
|
||||
output_modalities: z.array(z.string()),
|
||||
tokenizer: z.string().optional(),
|
||||
}),
|
||||
pricing: z.object({
|
||||
prompt: z.string(),
|
||||
completion: z.string(),
|
||||
internal_reasoning: z.string().optional(),
|
||||
input_cache_read: z.string().optional(),
|
||||
input_cache_write: z.string().optional(),
|
||||
}),
|
||||
top_provider: z.object({
|
||||
context_length: z.number().nullable(),
|
||||
max_completion_tokens: z.number().nullable(),
|
||||
is_moderated: z.boolean().optional(),
|
||||
}),
|
||||
supported_parameters: z.array(z.string()),
|
||||
opencode: z
|
||||
.object({
|
||||
variants: z
|
||||
.record(
|
||||
z.object({
|
||||
reasoning: z
|
||||
.object({
|
||||
enabled: z.boolean(),
|
||||
effort: z.string().optional(),
|
||||
})
|
||||
.optional(),
|
||||
}),
|
||||
)
|
||||
.optional(),
|
||||
})
|
||||
.optional(),
|
||||
});
|
||||
|
||||
export const KiloResponse = z.object({
|
||||
data: z.array(KiloModel),
|
||||
}).passthrough();
|
||||
|
||||
export type KiloModel = z.infer<typeof KiloModel>;
|
||||
|
||||
export const kilo = {
|
||||
id: "kilo",
|
||||
name: "Kilo",
|
||||
modelsDir: "providers/kilo/models",
|
||||
async fetchModels() {
|
||||
const headers = process.env.KILO_API_KEY
|
||||
? { Authorization: `Bearer ${process.env.KILO_API_KEY}` }
|
||||
: undefined;
|
||||
const response = await fetch(API_ENDPOINT, { headers });
|
||||
if (!response.ok) {
|
||||
throw new Error(`Kilo request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return KiloResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
// Kilo serves deprecated/unavailable routes as degraded stubs:
|
||||
// negative pricing (`"-1"`) and an empty `supported_parameters` array. Syncing
|
||||
// those would wrongly flip `reasoning`/`tool_call`/`structured_output` to false
|
||||
// and strip `reasoning_options`. Leave the authored file untouched instead, and
|
||||
// skip the model entirely when we have nothing to preserve.
|
||||
if (isUnavailable(model)) {
|
||||
const authored = context.authored(model.id);
|
||||
return authored === undefined ? undefined : { id: model.id, model: authored as SyncedModel };
|
||||
}
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildKiloModel(model, context.existing(model.id)),
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<KiloModel>;
|
||||
|
||||
function isUnavailable(model: KiloModel) {
|
||||
return (
|
||||
model.supported_parameters.length === 0 ||
|
||||
Number(model.pricing.prompt) < 0 ||
|
||||
Number(model.pricing.completion) < 0
|
||||
);
|
||||
}
|
||||
|
||||
function dateFromTimestamp(timestamp: number) {
|
||||
return new Date(timestamp * 1000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function price(value: string | undefined) {
|
||||
if (value === undefined) return undefined;
|
||||
const number = Number(value);
|
||||
return Number.isFinite(number) && number >= 0
|
||||
? Math.round(number * 1_000_000_000_000) / 1_000_000
|
||||
: undefined;
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[], fallback: Modality[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => value.toLowerCase())
|
||||
.map((value) => value === "file" ? "pdf" : value)
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : fallback)];
|
||||
}
|
||||
|
||||
function inferFamily(model: KiloModel, name: string) {
|
||||
const kimiFamily = inferKimiFamily(model.id, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const target = `${model.id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
.find((family) => {
|
||||
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
if (family === "o") {
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=\\d|$|[^a-z0-9])`).test(target);
|
||||
}
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
|
||||
});
|
||||
}
|
||||
|
||||
export function buildKiloModel(
|
||||
model: KiloModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel?: string,
|
||||
): SyncedModel {
|
||||
const params = new Set(model.supported_parameters);
|
||||
const name = model.name;
|
||||
const apiDescription = model.description?.replaceAll(/\s+/g, " ").trim();
|
||||
const input = modalities(model.architecture.input_modalities, ["text"]);
|
||||
const output = modalities(model.architecture.output_modalities, ["text"]);
|
||||
const prompt = price(model.pricing.prompt);
|
||||
const completion = price(model.pricing.completion);
|
||||
const reasoning = params.has("reasoning") || params.has("include_reasoning");
|
||||
const reasoning_options = existing?.reasoning_options?.length
|
||||
? existing.reasoning_options
|
||||
: KiloReasoningOptions(model.opencode) ?? existing?.reasoning_options;
|
||||
const context = model.top_provider.context_length ?? model.context_length;
|
||||
const family = inferFamily(model, name);
|
||||
const releaseDate = dateFromTimestamp(model.created);
|
||||
const familyValue = existing?.family === "o" && family !== "o"
|
||||
? family
|
||||
: (existing?.family ?? family);
|
||||
const attachment = input.some((value) => value !== "text");
|
||||
const toolCall = params.has("tools") || params.has("tool_choice");
|
||||
const structuredOutput = params.has("structured_outputs");
|
||||
const knowledge = model.knowledge_cutoff?.slice(0, 10) ?? existing?.knowledge;
|
||||
const openWeights = Boolean(model.hugging_face_id);
|
||||
const cost = prompt !== undefined && completion !== undefined
|
||||
? {
|
||||
input: prompt,
|
||||
output: completion,
|
||||
reasoning: reasoning ? price(model.pricing.internal_reasoning) : undefined,
|
||||
cache_read: price(model.pricing.input_cache_read),
|
||||
cache_write: price(model.pricing.input_cache_write),
|
||||
tiers: existing?.cost?.tiers,
|
||||
}
|
||||
: existing?.cost;
|
||||
const limit = {
|
||||
context,
|
||||
input: existing?.limit?.input,
|
||||
output: model.top_provider.max_completion_tokens ?? existing?.limit?.output ?? context,
|
||||
};
|
||||
const canonical = existing?.base_model ?? baseModel ?? resolveCanonicalBaseModel(model.id);
|
||||
|
||||
if (canonical !== undefined) {
|
||||
return factorBaseModel(
|
||||
canonical,
|
||||
{
|
||||
name: baseModel !== undefined || model.id.endsWith(":free") ? name : undefined,
|
||||
description: existing?.description ?? apiDescription ?? describeModel({
|
||||
id: model.id,
|
||||
name,
|
||||
family: familyValue,
|
||||
reasoning,
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput,
|
||||
open_weights: openWeights,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
}),
|
||||
attachment,
|
||||
reasoning,
|
||||
reasoning_options,
|
||||
temperature: params.has("temperature"),
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
cost,
|
||||
},
|
||||
limit,
|
||||
existing?.base_model === canonical ? existing.base_model_omit : undefined,
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
name,
|
||||
description: existing?.description ?? apiDescription ?? describeModel({
|
||||
id: model.id,
|
||||
name,
|
||||
family: familyValue,
|
||||
reasoning,
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput,
|
||||
open_weights: openWeights,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
}),
|
||||
family: familyValue,
|
||||
release_date: releaseDate,
|
||||
last_updated: releaseDate,
|
||||
attachment,
|
||||
reasoning,
|
||||
reasoning_options,
|
||||
temperature: params.has("temperature"),
|
||||
tool_call: toolCall,
|
||||
structured_output: structuredOutput,
|
||||
knowledge,
|
||||
open_weights: openWeights,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
} satisfies SyncedFullModel;
|
||||
}
|
||||
|
||||
function KiloReasoningOptions(opencode: KiloModel["opencode"]): SyncedFullModel["reasoning_options"] {
|
||||
if (opencode?.variants === undefined) return undefined;
|
||||
|
||||
const options: NonNullable<SyncedFullModel["reasoning_options"]> = [];
|
||||
const variants = Object.entries(opencode.variants);
|
||||
|
||||
if (variants.length === 0) return undefined;
|
||||
|
||||
const reasoningEffortOrder = new Map<string, number>([
|
||||
["none", 0],
|
||||
["minimal", 1],
|
||||
["low", 2],
|
||||
["medium", 3],
|
||||
["high", 4],
|
||||
["xhigh", 5],
|
||||
["max", 6],
|
||||
]);
|
||||
|
||||
const efforts = variants
|
||||
.filter(([, variant]) => variant.reasoning?.enabled === true)
|
||||
.map(([, variant]) => variant.reasoning?.effort)
|
||||
.filter((effort): effort is string => effort !== undefined);
|
||||
const hasNone = variants.some(([, variant]) => variant.reasoning?.enabled === false);
|
||||
const allEfforts = hasNone ? [...efforts, "none"] : [...efforts];
|
||||
|
||||
if (allEfforts.length > 0) {
|
||||
const orderedEfforts = allEfforts.sort((a, b) => {
|
||||
const order = (reasoningEffortOrder.get(a) ?? Number.MAX_SAFE_INTEGER)
|
||||
- (reasoningEffortOrder.get(b) ?? Number.MAX_SAFE_INTEGER);
|
||||
return order;
|
||||
});
|
||||
options.push({
|
||||
type: "effort",
|
||||
values: orderedEfforts as Array<string | null>,
|
||||
});
|
||||
}
|
||||
|
||||
return options.length > 0 ? options : undefined;
|
||||
}
|
||||
|
||||
function modelMetadataExists(provider: string, modelID: string) {
|
||||
let files = modelMetadataFilesByProvider.get(provider);
|
||||
if (files === undefined) {
|
||||
try {
|
||||
files = new Set(readdirSync(path.join(MODELS_DIR, provider)));
|
||||
} catch {
|
||||
files = new Set();
|
||||
}
|
||||
modelMetadataFilesByProvider.set(provider, files);
|
||||
}
|
||||
return files.has(`${modelID}.toml`);
|
||||
}
|
||||
|
||||
function baseModelOmit(
|
||||
modelID: string,
|
||||
limit: SyncedFullModel["limit"],
|
||||
) {
|
||||
const metadata = modelMetadata(modelID);
|
||||
const omit: string[] = [];
|
||||
const baseLimit = metadata.limit;
|
||||
if (
|
||||
isPlainObject(baseLimit) &&
|
||||
baseLimit.input !== undefined &&
|
||||
limit.input === undefined &&
|
||||
baseLimit.context !== limit.context
|
||||
) {
|
||||
omit.push("limit.input");
|
||||
}
|
||||
|
||||
return omit.length > 0 ? omit : undefined;
|
||||
}
|
||||
|
||||
function baseModelOverrides(
|
||||
modelID: string,
|
||||
values: Partial<SyncedFullModel>,
|
||||
) {
|
||||
const metadata = modelMetadata(modelID);
|
||||
const result: Record<string, unknown> = {};
|
||||
|
||||
for (const [key, value] of Object.entries(values)) {
|
||||
const override = inheritedOverride(value, metadata[key]);
|
||||
if (override !== undefined) result[key] = override;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
function inheritedOverride(value: unknown, inherited: unknown): unknown {
|
||||
if (value === undefined) return undefined;
|
||||
if (sameInheritedValue(value, inherited)) return undefined;
|
||||
if (isPlainObject(value) && isPlainObject(inherited)) {
|
||||
const overrides = Object.fromEntries(
|
||||
Object.entries(value)
|
||||
.map(([key, item]) => [key, inheritedOverride(item, inherited[key])])
|
||||
.filter(([, item]) => item !== undefined),
|
||||
);
|
||||
return Object.keys(overrides).length > 0 ? overrides : undefined;
|
||||
}
|
||||
return stripUndefined(value);
|
||||
}
|
||||
|
||||
function stripUndefined(value: unknown): unknown {
|
||||
if (Array.isArray(value)) return value.map(stripUndefined);
|
||||
if (isPlainObject(value)) {
|
||||
return Object.fromEntries(
|
||||
Object.entries(value)
|
||||
.filter(([, item]) => item !== undefined)
|
||||
.map(([key, item]) => [key, stripUndefined(item)]),
|
||||
);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function sameInheritedValue(value: unknown, inherited: unknown) {
|
||||
return stableInheritedValue(value) === stableInheritedValue(inherited);
|
||||
}
|
||||
|
||||
function stableInheritedValue(value: unknown): string {
|
||||
if (Array.isArray(value)) {
|
||||
const items = value.map(stableInheritedValue);
|
||||
const ordered = value.every((item) => item === null || typeof item !== "object")
|
||||
? items.sort()
|
||||
: items;
|
||||
return `[${ordered.join(",")}]`;
|
||||
}
|
||||
if (isPlainObject(value)) {
|
||||
return `{${Object.entries(value)
|
||||
.filter(([, item]) => item !== undefined)
|
||||
.sort(([a], [b]) => a.localeCompare(b))
|
||||
.map(([key, item]) => `${JSON.stringify(key)}:${stableInheritedValue(item)}`)
|
||||
.join(",")}}`;
|
||||
}
|
||||
return JSON.stringify(value);
|
||||
}
|
||||
|
||||
function isPlainObject(value: unknown): value is Record<string, unknown> {
|
||||
return value !== null && typeof value === "object" && !Array.isArray(value);
|
||||
}
|
||||
|
||||
function modelMetadata(modelID: string) {
|
||||
let metadata = modelMetadataByID.get(modelID);
|
||||
if (metadata === undefined) {
|
||||
const filePath = path.join(MODELS_DIR, `${modelID}.toml`);
|
||||
metadata = Bun.TOML.parse(readFileSync(filePath, "utf8")) as Record<string, unknown>;
|
||||
modelMetadataByID.set(modelID, metadata);
|
||||
}
|
||||
return metadata;
|
||||
}
|
||||
|
||||
function canonicalCandidates(provider: string, modelID: string) {
|
||||
const candidates = [modelID];
|
||||
|
||||
if (provider === "anthropic") {
|
||||
candidates.push(modelID.replace(/(claude-(?:opus|sonnet|haiku)-\d+)\.(\d+)/, "$1-$2"));
|
||||
candidates.push(modelID.replace(/^claude-3\.5-/, "claude-3-5-"));
|
||||
}
|
||||
|
||||
if (provider === "llama") {
|
||||
candidates.push(modelID.replace(/^llama-(\d+)-(\d+)/, "llama-$1.$2"));
|
||||
candidates.push(modelID.replace(/^llama-(4)-(maverick|scout)$/, "llama-$1-$2-17b"));
|
||||
}
|
||||
|
||||
if (provider === "mistral") {
|
||||
candidates.push(modelID.replace(/-latest$/, ""));
|
||||
}
|
||||
|
||||
if (provider === "minimax") {
|
||||
candidates.push(modelID.replace(/^minimax-m/, "MiniMax-M"));
|
||||
}
|
||||
|
||||
return [...new Set(candidates)];
|
||||
}
|
||||
@@ -0,0 +1,339 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import { describeModel } from "../../describe.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveCanonicalBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api-gateway.merge.dev/v1/models";
|
||||
|
||||
const AvailabilityStatus = z.enum(["available", "deprecated"]);
|
||||
|
||||
const VendorReasoning = z.object({
|
||||
configurable: z.boolean().optional(),
|
||||
disable_supported: z.boolean().optional(),
|
||||
default_enabled: z.boolean().optional(),
|
||||
controls: z.array(z.string()).optional(),
|
||||
output_style: z.string().nullable().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const VendorCapabilities = z.object({
|
||||
// Keep the API boundary forward-compatible; `modalities()` filters the
|
||||
// evolving Gateway vocabulary to values supported by models.dev.
|
||||
input: z.array(z.string()),
|
||||
output: z.array(z.string()),
|
||||
supports_tool_calling: z.boolean(),
|
||||
supports_tool_choice: z.boolean().default(false),
|
||||
supports_structured_outputs: z.boolean(),
|
||||
supports_reasoning: z.boolean().optional(),
|
||||
reasoning: VendorReasoning.nullable().optional(),
|
||||
streaming: z.boolean(),
|
||||
}).passthrough();
|
||||
|
||||
const PromptCaching = z.object({
|
||||
mode: z.enum(["automatic", "explicit", "none"]).optional(),
|
||||
cache_read_cost_per_million: z.number().nonnegative().nullable().optional(),
|
||||
cache_write_cost_per_million: z.number().nonnegative().nullable().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const VendorInfo = z.object({
|
||||
launch_date: z.string().nullable().optional(),
|
||||
context_window: z.number().int().nonnegative(),
|
||||
max_output_tokens: z.number().int().nonnegative(),
|
||||
availability_status: AvailabilityStatus,
|
||||
capabilities: VendorCapabilities,
|
||||
pricing: z.object({
|
||||
currency: z.literal("USD").default("USD"),
|
||||
input_per_million: z.number().nonnegative(),
|
||||
output_per_million: z.number().nonnegative(),
|
||||
cache_read_per_million: z.number().nonnegative().nullable().optional(),
|
||||
cache_write_per_million: z.number().nonnegative().nullable().optional(),
|
||||
}).passthrough(),
|
||||
prompt_caching: PromptCaching.nullable().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const MergeGatewayModel = z.object({
|
||||
model: z.string().min(1),
|
||||
provider: z.string().min(1),
|
||||
display_name: z.string().min(1),
|
||||
vendors: z.record(VendorInfo),
|
||||
availability_status: AvailabilityStatus,
|
||||
created_at: z.string().nullable().optional(),
|
||||
updated_at: z.string().nullable().optional(),
|
||||
}).passthrough().superRefine((model, context) => {
|
||||
const namespace = model.model.split("/")[0];
|
||||
if (namespace !== model.provider) {
|
||||
context.addIssue({
|
||||
code: z.ZodIssueCode.custom,
|
||||
path: ["provider"],
|
||||
message: `Model namespace ${namespace} does not match provider ${model.provider}`,
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
export const MergeGatewayResponse = z.object({
|
||||
object: z.literal("list").default("list"),
|
||||
data: z.array(MergeGatewayModel),
|
||||
has_more: z.boolean().default(false),
|
||||
next_cursor: z.string().nullable().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export type MergeGatewayModel = z.infer<typeof MergeGatewayModel>;
|
||||
export type MergeGatewayVendor = z.infer<typeof VendorInfo>;
|
||||
|
||||
export async function fetchMergeGatewayModels(
|
||||
fetcher: typeof fetch = fetch,
|
||||
apiKey = process.env.MERGE_GATEWAY_API_KEY,
|
||||
) {
|
||||
if (!apiKey) throw new Error("MERGE_GATEWAY_API_KEY is required to sync Merge Gateway models");
|
||||
|
||||
const models = new Map<string, MergeGatewayModel>();
|
||||
const cursors = new Set<string>();
|
||||
let cursor: string | undefined;
|
||||
|
||||
do {
|
||||
const url = new URL(API_ENDPOINT);
|
||||
url.searchParams.set("limit", "500");
|
||||
if (cursor !== undefined) url.searchParams.set("cursor", cursor);
|
||||
|
||||
const response = await fetcher(url, {
|
||||
headers: { Authorization: `Bearer ${apiKey}` },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`Merge Gateway request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const page = MergeGatewayResponse.parse(await response.json());
|
||||
for (const model of page.data) {
|
||||
if (models.has(model.model)) {
|
||||
throw new Error(`Merge Gateway returned duplicate model ID: ${model.model}`);
|
||||
}
|
||||
models.set(model.model, model);
|
||||
}
|
||||
if (!page.has_more) break;
|
||||
if (!page.next_cursor) throw new Error("Merge Gateway returned has_more=true without next_cursor");
|
||||
if (cursors.has(page.next_cursor)) throw new Error(`Merge Gateway repeated cursor: ${page.next_cursor}`);
|
||||
cursors.add(page.next_cursor);
|
||||
cursor = page.next_cursor;
|
||||
} while (true);
|
||||
|
||||
return {
|
||||
object: "list" as const,
|
||||
data: [...models.values()],
|
||||
has_more: false,
|
||||
next_cursor: null,
|
||||
};
|
||||
}
|
||||
|
||||
export const mergeGateway = {
|
||||
id: "merge-gateway",
|
||||
name: "Merge Gateway",
|
||||
modelsDir: "providers/merge-gateway/models",
|
||||
// API-key policy can affect catalog visibility. Retain missing local models
|
||||
// until Merge exposes an account-independent catalog response.
|
||||
deleteMissing: false,
|
||||
sourceID(model) {
|
||||
return model.model;
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} Merge Gateway models were skipped because they are not text models or lack canonical metadata.`,
|
||||
`Skipped remote IDs: ${ids.map((id) => `\`${id}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
missingNotice(paths) {
|
||||
if (paths.length === 0) return [];
|
||||
return [
|
||||
`${paths.length} local Merge Gateway models were absent from the API response and retained for manual lifecycle review.`,
|
||||
`Retained local paths: ${paths.map((item) => `\`${item}\``).join(", ")}`,
|
||||
];
|
||||
},
|
||||
fetchModels() {
|
||||
return fetchMergeGatewayModels();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return MergeGatewayResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const existing = context.existing(model.model);
|
||||
const translated = buildMergeGatewayModel(model, existing, context.authored(model.model));
|
||||
return translated === undefined ? undefined : { id: model.model, model: translated };
|
||||
},
|
||||
} satisfies SyncProvider<MergeGatewayModel>;
|
||||
|
||||
export function selectMergeGatewayVendor(model: MergeGatewayModel) {
|
||||
const canonical = model.vendors[model.provider];
|
||||
if (canonical?.availability_status === "available") {
|
||||
return { id: model.provider, info: canonical };
|
||||
}
|
||||
|
||||
// Match Gateway's default resolver: when the model author's native route is
|
||||
// unavailable, use the cheapest active route by combined input + output
|
||||
// price. Object order is preserved for equal prices; the public API emits
|
||||
// vendors in CMS-priority order, which is Gateway's own tiebreaker.
|
||||
const available = Object.entries(model.vendors)
|
||||
.filter(([, info]) => info.availability_status === "available");
|
||||
const selected = available.reduce<typeof available[number] | undefined>((best, candidate) => {
|
||||
if (best === undefined) return candidate;
|
||||
const bestCost = best[1].pricing.input_per_million + best[1].pricing.output_per_million;
|
||||
const candidateCost = candidate[1].pricing.input_per_million + candidate[1].pricing.output_per_million;
|
||||
return candidateCost < bestCost ? candidate : best;
|
||||
}, undefined);
|
||||
if (selected !== undefined) return { id: selected[0], info: selected[1] };
|
||||
if (canonical !== undefined) return { id: model.provider, info: canonical };
|
||||
|
||||
const fallback = Object.entries(model.vendors)[0];
|
||||
return fallback === undefined ? undefined : { id: fallback[0], info: fallback[1] };
|
||||
}
|
||||
|
||||
export function buildMergeGatewayModel(
|
||||
model: MergeGatewayModel,
|
||||
existing: ExistingModel | undefined,
|
||||
authored: ExistingModel | undefined = existing,
|
||||
): SyncedModel | undefined {
|
||||
const selected = selectMergeGatewayVendor(model);
|
||||
if (selected === undefined || !selected.info.capabilities.output.includes("text")) return undefined;
|
||||
|
||||
const input = modalities(selected.info.capabilities.input);
|
||||
const output = modalities(selected.info.capabilities.output);
|
||||
const limit = {
|
||||
context: selected.info.context_window || existing?.limit?.context || 0,
|
||||
// Preserve only a provider-authored input cap. `existing` is resolved
|
||||
// against base-model metadata, so using its inherited input value here
|
||||
// can keep an impossible cap when the gateway reports a smaller context.
|
||||
input: authored?.limit?.input,
|
||||
output: selected.info.max_output_tokens || existing?.limit?.output || selected.info.context_window,
|
||||
};
|
||||
const cachePricing = mergeGatewayCachePricing(selected.info, existing);
|
||||
const cost = {
|
||||
input: selected.info.pricing.input_per_million,
|
||||
output: selected.info.pricing.output_per_million,
|
||||
reasoning: existing?.cost?.reasoning,
|
||||
cache_read: cachePricing.read,
|
||||
cache_write: cachePricing.write,
|
||||
input_audio: existing?.cost?.input_audio,
|
||||
output_audio: existing?.cost?.output_audio,
|
||||
tiers: existing?.cost?.tiers,
|
||||
};
|
||||
const status = model.availability_status === "deprecated" || selected.info.availability_status === "deprecated"
|
||||
? "deprecated" as const
|
||||
: undefined;
|
||||
const baseModel = existing?.base_model ?? resolveCanonicalBaseModel(model.model);
|
||||
// `supports_reasoning` is not part of the documented public schema
|
||||
// (PublicVendorModelCapabilities) and is inconsistently populated across
|
||||
// vendor routes: the same model can report `true` on one route and `false`
|
||||
// on another (e.g. anthropic/claude-opus-4-6 reports `false` via `anthropic`
|
||||
// and `true` via `bedrock`), and reasoning-only models such as
|
||||
// deepseek/deepseek-r1 report `false` on their sole route. Treat it as a
|
||||
// positive-only signal: `true` (always accompanied by route `reasoning`
|
||||
// metadata) confirms the model reasons on the gateway, while `false`/absent
|
||||
// means unknown and preserves curated reasoning metadata.
|
||||
const routeConfirmsReasoning = Object.values(model.vendors).some(
|
||||
(vendor) => vendor.availability_status === "available" && vendor.capabilities.supports_reasoning === true,
|
||||
);
|
||||
const reasoning = routeConfirmsReasoning ? true : existing?.reasoning;
|
||||
const existingReasoningOptions = existing?.reasoning_options ?? [];
|
||||
const reasoningOptions = reasoning === true && existingReasoningOptions.length === 0
|
||||
&& selected.info.capabilities.reasoning?.disable_supported === true
|
||||
? [{ type: "toggle" as const }]
|
||||
: reasoning === true
|
||||
? existingReasoningOptions
|
||||
: existing?.reasoning_options;
|
||||
const modelSlug = model.model.split("/").at(-1)?.toLowerCase();
|
||||
const displayNameIsID = model.display_name.includes("/")
|
||||
|| model.display_name.toLowerCase() === modelSlug;
|
||||
const authoritative = {
|
||||
// Some catalog rows use an upstream org/model ID as display_name. Let
|
||||
// canonical metadata provide the human-readable name for factored models.
|
||||
name: baseModel !== undefined && displayNameIsID ? undefined : model.display_name,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
tool_call: selected.info.capabilities.supports_tool_calling,
|
||||
structured_output: selected.info.capabilities.supports_structured_outputs,
|
||||
status,
|
||||
cost,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
};
|
||||
|
||||
if (baseModel !== undefined) {
|
||||
return factorBaseModel(
|
||||
baseModel,
|
||||
{
|
||||
...authoritative,
|
||||
description: existing?.description,
|
||||
reasoning,
|
||||
reasoning_options: reasoningOptions,
|
||||
temperature: existing?.temperature,
|
||||
interleaved: existing?.interleaved,
|
||||
provider: existing?.provider,
|
||||
experimental: existing?.experimental,
|
||||
},
|
||||
limit,
|
||||
existing?.base_model_omit,
|
||||
);
|
||||
}
|
||||
|
||||
if (existing === undefined) return undefined;
|
||||
|
||||
const releaseDate = selected.info.launch_date
|
||||
?? model.created_at?.slice(0, 10)
|
||||
?? existing.release_date;
|
||||
if (releaseDate === undefined) return undefined;
|
||||
const lastUpdated = model.updated_at?.slice(0, 10)
|
||||
?? existing.last_updated
|
||||
?? releaseDate;
|
||||
return {
|
||||
...authoritative,
|
||||
description: existing.description ?? describeModel({
|
||||
id: model.model,
|
||||
name: model.display_name,
|
||||
family: existing.family,
|
||||
reasoning,
|
||||
tool_call: selected.info.capabilities.supports_tool_calling,
|
||||
structured_output: selected.info.capabilities.supports_structured_outputs,
|
||||
open_weights: existing.open_weights,
|
||||
limit,
|
||||
modalities: { input, output },
|
||||
}),
|
||||
family: existing.family,
|
||||
release_date: releaseDate,
|
||||
last_updated: lastUpdated,
|
||||
reasoning: reasoning ?? false,
|
||||
reasoning_options: reasoningOptions,
|
||||
temperature: existing.temperature,
|
||||
knowledge: existing.knowledge,
|
||||
open_weights: existing.open_weights ?? false,
|
||||
interleaved: existing.interleaved,
|
||||
provider: existing.provider,
|
||||
experimental: existing.experimental,
|
||||
} satisfies SyncedFullModel;
|
||||
}
|
||||
|
||||
function mergeGatewayCachePricing(
|
||||
vendor: MergeGatewayVendor,
|
||||
existing: ExistingModel | undefined,
|
||||
) {
|
||||
const promptCaching = vendor.prompt_caching;
|
||||
const pricing = vendor.pricing;
|
||||
if (promptCaching?.mode === "none") {
|
||||
return { read: undefined, write: undefined };
|
||||
}
|
||||
return {
|
||||
read: promptCaching?.cache_read_cost_per_million
|
||||
?? pricing.cache_read_per_million
|
||||
?? existing?.cost?.cache_read,
|
||||
write: promptCaching?.cache_write_cost_per_million
|
||||
?? pricing.cache_write_per_million
|
||||
?? existing?.cost?.cache_write,
|
||||
};
|
||||
}
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
function modalities(values: string[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
return [...new Set(values
|
||||
.map((value) => value === "document" ? "pdf" : value)
|
||||
.filter((value): value is Modality => allowed.has(value as Modality))
|
||||
)];
|
||||
}
|
||||
@@ -0,0 +1,348 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import { inferKimiFamily, ModelFamilyValues } from "../../family.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel, resolveModelMetadataBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://nano-gpt.com/api/v1/models?detailed=true";
|
||||
|
||||
// NanoGPT accepts these exact request values, including `max`:
|
||||
// https://github.com/Nano-GPT-com/nanogpt/blob/073b25b07e9af619333c679e694de664bf1ceb30/lib/utils/reasoningInput.ts#L12-L28
|
||||
const ReasoningEffort = z.enum(["none", "minimal", "low", "medium", "high", "xhigh", "max"]);
|
||||
|
||||
const Pricing = z.object({
|
||||
prompt: z.number().nullish(),
|
||||
completion: z.number().nullish(),
|
||||
input: z.number().nullish(),
|
||||
output: z.number().nullish(),
|
||||
cacheReadInputPer1kTokens: z.number().nullish(),
|
||||
cacheWriteInputPer1kTokens: z.number().nullish(),
|
||||
note: z.string().optional(),
|
||||
}).passthrough();
|
||||
|
||||
const Architecture = z.object({
|
||||
input_modalities: z.array(z.string()).optional(),
|
||||
output_modalities: z.array(z.string()).optional(),
|
||||
}).passthrough();
|
||||
|
||||
const Capabilities = z.object({
|
||||
vision: z.boolean().optional(),
|
||||
video_input: z.boolean().optional(),
|
||||
audio_input: z.boolean().optional(),
|
||||
reasoning: z.boolean().optional(),
|
||||
tool_calling: z.boolean().optional(),
|
||||
structured_output: z.boolean().optional(),
|
||||
pdf_upload: z.boolean().optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const NanoGptModel = z.object({
|
||||
id: z.string().min(1),
|
||||
name: z.string().nullish(),
|
||||
description: z.string().nullish(),
|
||||
created: z.number().nullish(),
|
||||
owned_by: z.string().nullish(),
|
||||
context_length: z.number().int().nonnegative().nullish(),
|
||||
max_output_tokens: z.number().int().nonnegative().nullish(),
|
||||
architecture: Architecture.optional(),
|
||||
capabilities: Capabilities.optional(),
|
||||
reasoning_efforts: z.array(ReasoningEffort).nullish(),
|
||||
open_weights: z.boolean().nullish(),
|
||||
pricing: Pricing.optional(),
|
||||
}).passthrough();
|
||||
|
||||
export const NanoGptResponse = z.object({
|
||||
data: z.array(NanoGptModel),
|
||||
}).passthrough();
|
||||
|
||||
export type NanoGptModel = z.infer<typeof NanoGptModel>;
|
||||
|
||||
type Modality = "text" | "audio" | "image" | "video" | "pdf";
|
||||
|
||||
export const nanoGpt = {
|
||||
id: "nano-gpt",
|
||||
name: "NanoGPT",
|
||||
modelsDir: "providers/nano-gpt/models",
|
||||
preserveDescriptions: false,
|
||||
async fetchModels() {
|
||||
const response = await fetch(process.env.NANO_GPT_MODELS_URL ?? API_ENDPOINT);
|
||||
if (!response.ok) {
|
||||
throw new Error(`NanoGPT models request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return NanoGptResponse.parse(raw).data;
|
||||
},
|
||||
translateModel(model, context) {
|
||||
const id = normalizeModelID(model.id);
|
||||
const existing = context.existing(id);
|
||||
const baseModel = existing?.base_model ?? resolveNanoGptBaseModel(model.id);
|
||||
const translated = buildNanoGptModel(model, existing, baseModel);
|
||||
if (translated === undefined) return undefined;
|
||||
return {
|
||||
id,
|
||||
model: translated,
|
||||
};
|
||||
},
|
||||
} satisfies SyncProvider<NanoGptModel>;
|
||||
|
||||
const ORG_ID_NORMALIZATION: Record<string, string | undefined> = {
|
||||
nousresearch: "NousResearch",
|
||||
qwen: "qwen",
|
||||
thedrummer: "TheDrummer",
|
||||
};
|
||||
|
||||
const BASE_MODEL_ALIASES: Record<string, string | undefined> = {
|
||||
"claude-opus-4": "anthropic/claude-opus-4-0",
|
||||
"claude-sonnet-4": "anthropic/claude-sonnet-4-0",
|
||||
"cohere/north-mini-code": "cohere/north-mini-code-1-0",
|
||||
};
|
||||
|
||||
const NANO_GPT_VARIANT_SUFFIX = /(?::(?:thinking|none|minimal|low|medium|high|xhigh|max|\d+)|-thinking)$/i;
|
||||
|
||||
const KNOWN_OPEN_WEIGHT_IDS = new Set([
|
||||
"nex-agi/nex-n2-pro",
|
||||
]);
|
||||
|
||||
export function buildNanoGptModel(
|
||||
model: NanoGptModel,
|
||||
existing: ExistingModel | undefined,
|
||||
baseModel = existing?.base_model ?? resolveNanoGptBaseModel(model.id),
|
||||
): SyncedModel | undefined {
|
||||
const capabilities = model.capabilities ?? {};
|
||||
const explicitInputModalities = model.architecture?.input_modalities;
|
||||
const hasInputCapabilityMetadata = capabilities.vision !== undefined
|
||||
|| capabilities.audio_input !== undefined
|
||||
|| capabilities.video_input !== undefined
|
||||
|| capabilities.pdf_upload !== undefined;
|
||||
const addedInputModalities = [
|
||||
...(capabilities.vision ? ["image"] : []),
|
||||
...(capabilities.audio_input ? ["audio"] : []),
|
||||
...(capabilities.video_input ? ["video"] : []),
|
||||
...(capabilities.pdf_upload ? ["pdf"] : []),
|
||||
];
|
||||
const hasInputMetadata = explicitInputModalities !== undefined || hasInputCapabilityMetadata;
|
||||
const hasOutputMetadata = model.architecture?.output_modalities !== undefined;
|
||||
const input = normalizeModalities([
|
||||
...explicitInputModalities
|
||||
?? (hasInputCapabilityMetadata ? ["text"] : existing?.modalities?.input)
|
||||
?? ["text"],
|
||||
...addedInputModalities,
|
||||
]);
|
||||
const output = normalizeModalities(
|
||||
model.architecture?.output_modalities ?? existing?.modalities?.output ?? ["text"],
|
||||
);
|
||||
const sourceContext = positive(model.context_length);
|
||||
const sourceOutputLimit = positive(model.max_output_tokens);
|
||||
const context = sourceContext ?? existing?.limit?.context;
|
||||
const inputLimit = sourceContext ?? existing?.limit?.input;
|
||||
const outputLimit = sourceOutputLimit ?? existing?.limit?.output;
|
||||
const releaseDate = dateFromTimestamp(model.created) ?? existing?.release_date;
|
||||
const inferredSourceReasoning = capabilities.reasoning
|
||||
?? (model.reasoning_efforts != null ? true : undefined);
|
||||
const reasoning = inferredSourceReasoning ?? existing?.reasoning ?? false;
|
||||
const cost = buildCost(model.pricing, existing);
|
||||
if (baseModel !== undefined) {
|
||||
const existingAlreadyFactored = existing?.base_model === baseModel;
|
||||
const factoredModalities = {
|
||||
input: hasInputMetadata || existing !== undefined ? input : undefined,
|
||||
output: hasOutputMetadata || existing !== undefined ? output : undefined,
|
||||
};
|
||||
const factoredLimit = {
|
||||
context: sourceContext ?? existing?.limit?.context,
|
||||
input: sourceContext ?? existing?.limit?.input,
|
||||
output: sourceOutputLimit ?? existing?.limit?.output,
|
||||
};
|
||||
const sourceReasoning = inferredSourceReasoning;
|
||||
const sourceReasoningOptions = reasoningOptions(model, sourceReasoning, existing?.reasoning_options);
|
||||
|
||||
return factorBaseModel(
|
||||
baseModel,
|
||||
{
|
||||
name: existing?.name ?? model.name ?? undefined,
|
||||
description: existingAlreadyFactored ? existing?.description : undefined,
|
||||
family: existingAlreadyFactored ? existing?.family : undefined,
|
||||
release_date: existingAlreadyFactored ? existing?.release_date : undefined,
|
||||
last_updated: existingAlreadyFactored ? existing?.last_updated : undefined,
|
||||
attachment: hasInputMetadata
|
||||
? input.some((value) => value !== "text")
|
||||
: existing?.attachment,
|
||||
reasoning: sourceReasoning ?? existing?.reasoning,
|
||||
reasoning_options: sourceReasoningOptions,
|
||||
temperature: existing?.temperature,
|
||||
tool_call: capabilities.tool_calling ?? existing?.tool_call,
|
||||
structured_output: capabilities.structured_output ?? existing?.structured_output,
|
||||
knowledge: existing?.knowledge,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
provider: existing?.provider,
|
||||
experimental: existing?.experimental,
|
||||
cost,
|
||||
limit: factoredLimit,
|
||||
modalities: factoredModalities,
|
||||
},
|
||||
factoredLimit,
|
||||
existingAlreadyFactored ? existing?.base_model_omit : undefined,
|
||||
);
|
||||
}
|
||||
|
||||
if (context === undefined || outputLimit === undefined || releaseDate === undefined) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const values = {
|
||||
name: existing?.name ?? model.name ?? humanizeModelName(model.id),
|
||||
description: existing?.description ?? model.description ?? `${model.name ?? humanizeModelName(model.id)} on NanoGPT.`,
|
||||
family: existing?.family ?? inferFamily(model.id, model.name ?? ""),
|
||||
release_date: releaseDate,
|
||||
last_updated: existing?.last_updated ?? releaseDate,
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning,
|
||||
reasoning_options: reasoningOptions(model, reasoning, existing?.reasoning_options),
|
||||
temperature: existing?.temperature,
|
||||
tool_call: capabilities.tool_calling ?? existing?.tool_call ?? false,
|
||||
structured_output: capabilities.structured_output ?? existing?.structured_output,
|
||||
knowledge: existing?.knowledge,
|
||||
status: existing?.status,
|
||||
interleaved: existing?.interleaved,
|
||||
provider: existing?.provider,
|
||||
experimental: existing?.experimental,
|
||||
cost,
|
||||
limit: { context, input: inputLimit ?? context, output: outputLimit },
|
||||
modalities: { input, output },
|
||||
};
|
||||
|
||||
return {
|
||||
...values,
|
||||
open_weights: model.open_weights
|
||||
?? (KNOWN_OPEN_WEIGHT_IDS.has(model.id.toLowerCase()) ? true : existing?.open_weights)
|
||||
?? false,
|
||||
} satisfies SyncedFullModel;
|
||||
}
|
||||
|
||||
function buildCost(
|
||||
pricing: NanoGptModel["pricing"],
|
||||
existing: ExistingModel | undefined,
|
||||
): SyncedFullModel["cost"] {
|
||||
if (pricing === undefined) return existing?.cost;
|
||||
if (pricing.note === "varies_by_modality") return existing?.cost;
|
||||
|
||||
const input = pricing.input ?? pricing.prompt;
|
||||
const output = pricing.output ?? pricing.completion;
|
||||
if (!validPrice(input) || !validPrice(output)) return existing?.cost;
|
||||
|
||||
return {
|
||||
input: price(input),
|
||||
output: price(output),
|
||||
reasoning: existing?.cost?.reasoning,
|
||||
cache_read: !validPrice(pricing.cacheReadInputPer1kTokens)
|
||||
? existing?.cost?.cache_read
|
||||
: price(pricing.cacheReadInputPer1kTokens * 1_000),
|
||||
cache_write: !validPrice(pricing.cacheWriteInputPer1kTokens)
|
||||
? existing?.cost?.cache_write
|
||||
: price(pricing.cacheWriteInputPer1kTokens * 1_000),
|
||||
input_audio: existing?.cost?.input_audio,
|
||||
output_audio: existing?.cost?.output_audio,
|
||||
tiers: existing?.cost?.tiers,
|
||||
};
|
||||
}
|
||||
|
||||
function reasoningOptions(
|
||||
model: NanoGptModel,
|
||||
reasoning: boolean | undefined,
|
||||
existing: SyncedFullModel["reasoning_options"],
|
||||
): SyncedFullModel["reasoning_options"] {
|
||||
if (reasoning === false) return undefined;
|
||||
if (reasoning === undefined) return existing;
|
||||
if (model.reasoning_efforts == null) return existing ?? [];
|
||||
if (model.reasoning_efforts.length === 0) return [];
|
||||
return [{ type: "effort", values: [...model.reasoning_efforts] }];
|
||||
}
|
||||
|
||||
export function resolveNanoGptBaseModel(modelID: string) {
|
||||
let normalized = normalizeModelID(modelID);
|
||||
if (normalized.toLowerCase().startsWith("tee/")) {
|
||||
normalized = normalizeModelID(normalized.slice("TEE/".length));
|
||||
}
|
||||
|
||||
const exact = resolveNanoGptCanonicalCandidate(normalized);
|
||||
if (exact !== undefined) return exact;
|
||||
|
||||
const stripped = stripNanoGptVariantSuffixes(normalized);
|
||||
return stripped === normalized ? undefined : resolveNanoGptCanonicalCandidate(stripped);
|
||||
}
|
||||
|
||||
function resolveNanoGptCanonicalCandidate(modelID: string) {
|
||||
return BASE_MODEL_ALIASES[modelID.toLowerCase()] ?? resolveModelMetadataBaseModel(modelID);
|
||||
}
|
||||
|
||||
function stripNanoGptVariantSuffixes(modelID: string) {
|
||||
let normalized = modelID;
|
||||
while (true) {
|
||||
const stripped = normalized.replace(NANO_GPT_VARIANT_SUFFIX, "");
|
||||
if (stripped === normalized) return normalized;
|
||||
normalized = stripped;
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeModalities(values: string[]): Modality[] {
|
||||
const allowed = new Set<Modality>(["text", "audio", "image", "video", "pdf"]);
|
||||
const result = values
|
||||
.map((value) => normalizeModality(value))
|
||||
.filter((value): value is Modality => allowed.has(value as Modality));
|
||||
return [...new Set(result.length > 0 ? result : ["text"] as Modality[])];
|
||||
}
|
||||
|
||||
function normalizeModality(value: string) {
|
||||
const lower = value.toLowerCase();
|
||||
if (lower === "images") return "image";
|
||||
if (lower === "videos") return "video";
|
||||
if (lower === "audios") return "audio";
|
||||
if (lower === "documents") return "pdf";
|
||||
return lower;
|
||||
}
|
||||
|
||||
function normalizeModelID(modelId: string) {
|
||||
const [org, ...parts] = modelId.split("/");
|
||||
if (org === undefined || parts.length === 0) return modelId;
|
||||
const normalizedOrg = ORG_ID_NORMALIZATION[org.toLowerCase()];
|
||||
return normalizedOrg === undefined ? modelId : `${normalizedOrg}/${parts.join("/")}`;
|
||||
}
|
||||
|
||||
function inferFamily(id: string, name: string) {
|
||||
const kimiFamily = inferKimiFamily(id, name);
|
||||
if (kimiFamily !== undefined) return kimiFamily;
|
||||
|
||||
const target = `${id} ${name}`.toLowerCase();
|
||||
return [...ModelFamilyValues]
|
||||
.sort((a, b) => b.length - a.length)
|
||||
.find((family) => {
|
||||
const value = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
if (family === "o") return new RegExp(`(^|[^a-z0-9])${value}(?=\\d)`).test(target);
|
||||
return new RegExp(`(^|[^a-z0-9])${value}(?=$|[^a-z0-9])`).test(target);
|
||||
});
|
||||
}
|
||||
|
||||
function humanizeModelName(modelId: string) {
|
||||
const modelPart = modelId.split("/").at(-1) ?? modelId;
|
||||
return modelPart
|
||||
.replace(/[:/_-]+/g, " ")
|
||||
.replace(/\b\w/g, (value) => value.toUpperCase());
|
||||
}
|
||||
|
||||
function dateFromTimestamp(timestamp: number | null | undefined) {
|
||||
if (timestamp == null || timestamp <= 0) return undefined;
|
||||
return new Date(timestamp * 1_000).toISOString().slice(0, 10);
|
||||
}
|
||||
|
||||
function positive(value: number | null | undefined) {
|
||||
return value == null || value <= 0 ? undefined : value;
|
||||
}
|
||||
|
||||
function price(value: number) {
|
||||
return Math.round(value * 1_000_000) / 1_000_000;
|
||||
}
|
||||
|
||||
function validPrice(value: number | null | undefined): value is number {
|
||||
return value !== null && value !== undefined && value >= 0;
|
||||
}
|
||||
@@ -56,6 +56,9 @@ export const openai = {
|
||||
name: "OpenAI",
|
||||
modelsDir: "providers/openai/models",
|
||||
skipCreates: true,
|
||||
// /v1/models is account-scoped and includes legacy, internal, and
|
||||
// non-catalog surfaces without authoritative lifecycle metadata.
|
||||
trackMissingModels: false,
|
||||
deleteMissing: false,
|
||||
sourceID(model) {
|
||||
return model.id;
|
||||
|
||||
@@ -10,11 +10,14 @@ const API_ENDPOINT = "https://openrouter.ai/api/v1/models";
|
||||
const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models");
|
||||
const modelMetadataByID = new Map<string, Record<string, unknown>>();
|
||||
const modelMetadataFilesByProvider = new Map<string, Set<string>>();
|
||||
let allModelMetadataIDs: string[] | undefined;
|
||||
|
||||
const CANONICAL_BASE_MODEL_OVERRIDES = {
|
||||
"openai/gpt-5.6-luna-pro": "openai/gpt-5.6-luna",
|
||||
"openai/gpt-5.6-sol-pro": "openai/gpt-5.6-sol",
|
||||
"openai/gpt-5.6-terra-pro": "openai/gpt-5.6-terra",
|
||||
"anthropic/claude-opus-4.7-fast": "anthropic/claude-opus-4-7",
|
||||
"anthropic/claude-opus-4.8-fast": "anthropic/claude-opus-4-8",
|
||||
} as const;
|
||||
|
||||
const CANONICAL_PROVIDER_PREFIXES = {
|
||||
@@ -27,17 +30,22 @@ const CANONICAL_PROVIDER_PREFIXES = {
|
||||
"meta-llama": { provider: "llama", metadata: "meta" },
|
||||
minimax: { provider: "minimax", metadata: "minimax" },
|
||||
mistralai: { provider: "mistral", metadata: "mistral" },
|
||||
moonshot: { provider: "moonshotai", metadata: "moonshotai" },
|
||||
moonshotai: { provider: "moonshotai", metadata: "moonshotai" },
|
||||
openai: { provider: "openai", metadata: "openai" },
|
||||
nvidia: { provider: "nvidia", metadata: "nvidia" },
|
||||
qwen: { provider: "alibaba", metadata: "alibaba" },
|
||||
sakana: { provider: "sakana", metadata: "sakana" },
|
||||
stepfun: { provider: "stepfun", metadata: "stepfun" },
|
||||
"stepfun-ai": { provider: "stepfun", metadata: "stepfun" },
|
||||
tencent: { provider: "tencent", metadata: "tencent" },
|
||||
thinkingmachines: { provider: "thinkingmachines", metadata: "thinkingmachines" },
|
||||
"x-ai": { provider: "xai", metadata: "xai" },
|
||||
xai: { provider: "xai", metadata: "xai" },
|
||||
xiaomi: { provider: "xiaomi", metadata: "xiaomi" },
|
||||
zai: { provider: "zai", metadata: "zhipuai" },
|
||||
"z-ai": { provider: "zai", metadata: "zhipuai" },
|
||||
"zai-org": { provider: "zai", metadata: "zhipuai" },
|
||||
} as const;
|
||||
|
||||
export const OpenRouterModel = z.object({
|
||||
@@ -97,7 +105,8 @@ export const openrouter = {
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
return OpenRouterResponse.parse(raw).data;
|
||||
// Temporarily skip batch routes (`*:batch`) — they are not catalog targets.
|
||||
return OpenRouterResponse.parse(raw).data.filter((model) => !model.id.endsWith(":batch"));
|
||||
},
|
||||
translateModel(model, context) {
|
||||
// OpenRouter serves deprecated/unavailable routes as degraded stubs:
|
||||
@@ -175,9 +184,11 @@ export function buildOpenRouterModel(
|
||||
const prompt = price(model.pricing.prompt);
|
||||
const completion = price(model.pricing.completion);
|
||||
const reasoning = params.has("reasoning") || params.has("include_reasoning");
|
||||
const reasoning_options = existing?.reasoning_options?.length
|
||||
? existing.reasoning_options
|
||||
: openRouterReasoningOptions(model.reasoning) ?? existing?.reasoning_options;
|
||||
// Prefer OpenRouter's live reasoning metadata over authored options so aliases
|
||||
// and rotated models pick up new efforts/budget support. Fall back to authored
|
||||
// only when the API omits a reasoning object.
|
||||
const reasoning_options = openRouterReasoningOptions(model.reasoning)
|
||||
?? (reasoning ? existing?.reasoning_options : undefined);
|
||||
const context = model.context_length;
|
||||
const family = inferFamily(model, name);
|
||||
const releaseDate = dateFromTimestamp(model.created);
|
||||
@@ -211,7 +222,7 @@ export function buildOpenRouterModel(
|
||||
return factorBaseModel(
|
||||
canonical,
|
||||
{
|
||||
name: baseModel !== undefined || model.id.endsWith(":free") || canonicalOverride === canonical
|
||||
name: shouldPreserveFactoredName(model.id, canonical, baseModel, canonicalOverride)
|
||||
? name
|
||||
: undefined,
|
||||
description: existing?.description ?? describeModel({
|
||||
@@ -304,19 +315,39 @@ export function resolveCanonicalBaseModel(openrouterID: string) {
|
||||
if (prefix === undefined || modelParts.length === 0) return undefined;
|
||||
if (openrouterID.startsWith("~/") || prefix.startsWith("~")) return undefined;
|
||||
|
||||
const canonical = CANONICAL_PROVIDER_PREFIXES[prefix as keyof typeof CANONICAL_PROVIDER_PREFIXES];
|
||||
const canonical = CANONICAL_PROVIDER_PREFIXES[
|
||||
prefix.toLowerCase() as keyof typeof CANONICAL_PROVIDER_PREFIXES
|
||||
];
|
||||
if (canonical === undefined) return undefined;
|
||||
|
||||
const modelID = modelParts.join("/").replace(/:free$/, "");
|
||||
const candidates = canonicalCandidates(canonical.provider, modelID);
|
||||
const match = candidates.find((candidate) => {
|
||||
return modelMetadataExists(canonical.metadata, candidate);
|
||||
});
|
||||
const match = matchingModelMetadataFile(canonical.metadata, candidates);
|
||||
|
||||
return match === undefined ? undefined : `${canonical.metadata}/${match}`;
|
||||
}
|
||||
|
||||
function modelMetadataExists(provider: string, modelID: string) {
|
||||
/**
|
||||
* Resolve provider IDs that are not OpenRouter-shaped against the same canonical
|
||||
* metadata tree. Exact paths win; bare IDs only resolve when their filename is
|
||||
* unique across every metadata provider.
|
||||
*/
|
||||
export function resolveModelMetadataBaseModel(modelID: string) {
|
||||
const routed = resolveCanonicalBaseModel(modelID);
|
||||
if (routed !== undefined) return routed;
|
||||
|
||||
const normalized = modelID.replace(/:free$/, "");
|
||||
const ids = modelMetadataIDs();
|
||||
const exact = ids.find((candidate) => candidate.toLowerCase() === normalized.toLowerCase());
|
||||
if (exact !== undefined) return exact;
|
||||
if (normalized.includes("/")) return undefined;
|
||||
|
||||
const lower = normalized.toLowerCase();
|
||||
const matches = ids.filter((candidate) => candidate.split("/").at(-1)?.toLowerCase() === lower);
|
||||
return matches.length === 1 ? matches[0] : undefined;
|
||||
}
|
||||
|
||||
function matchingModelMetadataFile(provider: string, candidates: string[]) {
|
||||
let files = modelMetadataFilesByProvider.get(provider);
|
||||
if (files === undefined) {
|
||||
try {
|
||||
@@ -326,7 +357,30 @@ function modelMetadataExists(provider: string, modelID: string) {
|
||||
}
|
||||
modelMetadataFilesByProvider.set(provider, files);
|
||||
}
|
||||
return files.has(`${modelID}.toml`);
|
||||
|
||||
for (const candidate of candidates) {
|
||||
const expected = `${candidate}.toml`.toLowerCase();
|
||||
const match = [...files].find((file) => file.toLowerCase() === expected);
|
||||
if (match !== undefined) return match.slice(0, -".toml".length);
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function modelMetadataIDs() {
|
||||
if (allModelMetadataIDs !== undefined) return allModelMetadataIDs;
|
||||
|
||||
try {
|
||||
allModelMetadataIDs = readdirSync(MODELS_DIR, { withFileTypes: true })
|
||||
.filter((entry) => entry.isDirectory())
|
||||
.flatMap((entry) => {
|
||||
return readdirSync(path.join(MODELS_DIR, entry.name))
|
||||
.filter((file) => file.endsWith(".toml"))
|
||||
.map((file) => `${entry.name}/${file.slice(0, -".toml".length)}`);
|
||||
});
|
||||
} catch {
|
||||
allModelMetadataIDs = [];
|
||||
}
|
||||
return allModelMetadataIDs;
|
||||
}
|
||||
|
||||
function canonicalBaseModelOverride(openrouterID: string) {
|
||||
@@ -335,10 +389,36 @@ function canonicalBaseModelOverride(openrouterID: string) {
|
||||
];
|
||||
}
|
||||
|
||||
function shouldPreserveFactoredName(
|
||||
modelID: string,
|
||||
canonical: string,
|
||||
baseModel: string | undefined,
|
||||
canonicalOverride: string | undefined,
|
||||
) {
|
||||
if (baseModel !== undefined) return true;
|
||||
if (modelID.endsWith(":free")) return true;
|
||||
if (canonicalOverride === canonical) return true;
|
||||
const modelSlug = modelID.split("/").slice(1).join("/").replace(/:free$/, "");
|
||||
const canonicalSlug = canonical.split("/").slice(1).join("/");
|
||||
return normalizeModelSlug(modelSlug) !== normalizeModelSlug(canonicalSlug);
|
||||
}
|
||||
|
||||
function normalizeModelSlug(value: string) {
|
||||
return value.toLowerCase().replaceAll(/[^a-z0-9]/g, "");
|
||||
}
|
||||
|
||||
type BaseModelOverrides = Omit<Partial<SyncedFullModel>, "limit" | "modalities"> & {
|
||||
limit?: Partial<SyncedFullModel["limit"]>;
|
||||
modalities?: {
|
||||
input?: SyncedFullModel["modalities"]["input"];
|
||||
output?: SyncedFullModel["modalities"]["output"];
|
||||
};
|
||||
};
|
||||
|
||||
export function factorBaseModel(
|
||||
modelID: string,
|
||||
values: Partial<SyncedFullModel>,
|
||||
limit: SyncedFullModel["limit"],
|
||||
values: BaseModelOverrides,
|
||||
limit?: Partial<SyncedFullModel["limit"]>,
|
||||
existingOmit?: string[],
|
||||
): SyncedModel {
|
||||
return {
|
||||
@@ -350,14 +430,16 @@ export function factorBaseModel(
|
||||
|
||||
function baseModelOmit(
|
||||
modelID: string,
|
||||
limit: SyncedFullModel["limit"],
|
||||
limit: Partial<SyncedFullModel["limit"]> | undefined,
|
||||
) {
|
||||
if (limit === undefined) return undefined;
|
||||
const metadata = modelMetadata(modelID);
|
||||
const omit: string[] = [];
|
||||
const baseLimit = metadata.limit;
|
||||
if (
|
||||
isPlainObject(baseLimit) &&
|
||||
baseLimit.input !== undefined &&
|
||||
limit.context !== undefined &&
|
||||
limit.input === undefined &&
|
||||
baseLimit.context !== limit.context
|
||||
) {
|
||||
@@ -369,7 +451,7 @@ function baseModelOmit(
|
||||
|
||||
function baseModelOverrides(
|
||||
modelID: string,
|
||||
values: Partial<SyncedFullModel>,
|
||||
values: BaseModelOverrides,
|
||||
) {
|
||||
const metadata = modelMetadata(modelID);
|
||||
const result: Record<string, unknown> = {};
|
||||
@@ -446,10 +528,13 @@ function modelMetadata(modelID: string) {
|
||||
|
||||
function canonicalCandidates(provider: string, modelID: string) {
|
||||
const candidates = [modelID];
|
||||
if (modelID.endsWith("-fast")) candidates.push(modelID.slice(0, -"-fast".length));
|
||||
|
||||
if (provider === "anthropic") {
|
||||
candidates.push(modelID.replace(/(claude-(?:opus|sonnet|haiku)-\d+)\.(\d+)/, "$1-$2"));
|
||||
candidates.push(modelID.replace(/^claude-3\.5-/, "claude-3-5-"));
|
||||
for (const candidate of [...candidates]) {
|
||||
candidates.push(candidate.replace(/(claude-(?:opus|sonnet|haiku)-\d+)\.(\d+)/, "$1-$2"));
|
||||
candidates.push(candidate.replace(/^claude-3\.5-/, "claude-3-5-"));
|
||||
}
|
||||
}
|
||||
|
||||
if (provider === "llama") {
|
||||
|
||||
@@ -0,0 +1,232 @@
|
||||
import { z } from "zod";
|
||||
|
||||
import { describeModel } from "../../describe.js";
|
||||
import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js";
|
||||
import { factorBaseModel } from "./openrouter.js";
|
||||
|
||||
const API_ENDPOINT = "https://api.pioneer.ai/v1/models";
|
||||
|
||||
const BaseModels: Record<string, string> = {
|
||||
"Qwen/Qwen3.5-9B": "alibaba/qwen3.5-9b",
|
||||
"google/gemma-4-E2B-it": "google/gemma-4-E2B-it",
|
||||
"google/gemma-4-E4B-it": "google/gemma-4-E4B-it",
|
||||
"mistral-medium-3.5": "mistral/mistral-medium-2604",
|
||||
"moonshotai/Kimi-K2.7-Code": "moonshotai/kimi-k2.7-code",
|
||||
"openai/gpt-oss-120b": "openai/gpt-oss-120b",
|
||||
"openai/gpt-oss-20b": "openai/gpt-oss-20b",
|
||||
"sakana/fugu-ultra": "sakana/fugu-ultra",
|
||||
"zai-org/GLM-5.2": "zhipuai/glm-5.2",
|
||||
};
|
||||
|
||||
const Capability = z
|
||||
.object({
|
||||
supported: z.boolean(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const ReasoningEffortValues = [
|
||||
"none",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max",
|
||||
"default",
|
||||
] as const;
|
||||
|
||||
type ReasoningEffort = typeof ReasoningEffortValues[number];
|
||||
|
||||
const ReasoningEfforts = new Set<string>(ReasoningEffortValues);
|
||||
|
||||
const PioneerReasoningLevel = z
|
||||
.object({
|
||||
effort: z.string(),
|
||||
description: z.string().optional(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const PioneerMetadataModel = z
|
||||
.object({
|
||||
slug: z.string(),
|
||||
default_reasoning_level: z.string().nullish(),
|
||||
supported_reasoning_levels: z.array(PioneerReasoningLevel).nullish(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
const PioneerServedModel = z
|
||||
.object({
|
||||
id: z.string(),
|
||||
display_name: z.string(),
|
||||
created: z.number().optional(),
|
||||
created_at: z.string().optional(),
|
||||
max_input_tokens: z.number().int().nonnegative(),
|
||||
max_tokens: z.number().int().nonnegative(),
|
||||
deprecated: z.boolean().optional(),
|
||||
capabilities: z
|
||||
.object({
|
||||
image_input: Capability.optional(),
|
||||
pdf_input: Capability.optional(),
|
||||
structured_outputs: Capability.optional(),
|
||||
thinking: Capability.optional(),
|
||||
})
|
||||
.passthrough(),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
export const PioneerModel = PioneerServedModel.extend({
|
||||
metadata: PioneerMetadataModel.optional(),
|
||||
});
|
||||
|
||||
export const PioneerResponse = z
|
||||
.object({
|
||||
data: z.array(PioneerServedModel),
|
||||
models: z.array(PioneerMetadataModel).optional().default([]),
|
||||
})
|
||||
.passthrough();
|
||||
|
||||
export type PioneerModel = z.infer<typeof PioneerModel>;
|
||||
|
||||
export const pioneer = {
|
||||
id: "pioneer",
|
||||
name: "Pioneer",
|
||||
modelsDir: "providers/pioneer/models",
|
||||
skipCreates: true,
|
||||
// Pioneer reports 2024-01-01 for every model, so its creation dates cannot
|
||||
// support a meaningful age cutoff for remote-only model notifications.
|
||||
trackMissingModels: false,
|
||||
deleteMissing: false,
|
||||
async fetchModels() {
|
||||
const response = await fetch(API_ENDPOINT);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Pioneer request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return response.json();
|
||||
},
|
||||
parseModels(raw) {
|
||||
const parsed = PioneerResponse.parse(raw);
|
||||
const metadata = new Map(parsed.models.map((model) => [model.slug, model]));
|
||||
return parsed.data.map((model) => ({
|
||||
...model,
|
||||
metadata: metadata.get(model.id),
|
||||
}));
|
||||
},
|
||||
translateModel(model, context) {
|
||||
return {
|
||||
id: model.id,
|
||||
model: buildPioneerModel(model, context.existing(model.id)),
|
||||
};
|
||||
},
|
||||
missingNotice(paths) {
|
||||
if (paths.length === 0) return [];
|
||||
return [
|
||||
`${paths.length} local model(s) are not present in Pioneer /v1/models and were retained: ${paths.join(", ")}`,
|
||||
];
|
||||
},
|
||||
skippedNotice(ids) {
|
||||
if (ids.length === 0) return [];
|
||||
return [
|
||||
`${ids.length} remote model(s) are present in Pioneer /v1/models but were not created because Pioneer sync is update-only for new models: ${ids.join(", ")}`,
|
||||
];
|
||||
},
|
||||
} satisfies SyncProvider<PioneerModel>;
|
||||
|
||||
function dateFromModel(model: PioneerModel) {
|
||||
if (model.created !== undefined) return new Date(model.created * 1000).toISOString().slice(0, 10);
|
||||
if (model.created_at !== undefined) return model.created_at.slice(0, 10);
|
||||
return "2024-01-01";
|
||||
}
|
||||
|
||||
function supported(model: PioneerModel, capability: keyof PioneerModel["capabilities"]) {
|
||||
return model.capabilities[capability]?.supported === true;
|
||||
}
|
||||
|
||||
function isReasoningEffort(value: string): value is ReasoningEffort {
|
||||
return ReasoningEfforts.has(value);
|
||||
}
|
||||
|
||||
function pioneerReasoningOptions(model: PioneerModel): SyncedFullModel["reasoning_options"] {
|
||||
const levels = model.metadata?.supported_reasoning_levels ?? [];
|
||||
if (levels.length === 0) return undefined;
|
||||
|
||||
const unsupported = levels
|
||||
.map((level) => level.effort)
|
||||
.filter((effort) => !isReasoningEffort(effort));
|
||||
if (unsupported.length > 0) {
|
||||
throw new Error(
|
||||
`Unsupported Pioneer reasoning effort(s) for ${model.id}: ${[...new Set(unsupported)].join(", ")}`,
|
||||
);
|
||||
}
|
||||
|
||||
const values = [...new Set(levels.map((level) => level.effort).filter(isReasoningEffort))];
|
||||
return values.length > 0 ? [{ type: "effort", values }] : undefined;
|
||||
}
|
||||
|
||||
function buildPioneerModel(
|
||||
model: PioneerModel,
|
||||
existing: ExistingModel | undefined,
|
||||
): SyncedModel {
|
||||
const status = model.deprecated === true ? "deprecated" : existing?.status;
|
||||
const baseModel = existing?.base_model ?? BaseModels[model.id];
|
||||
const apiReasoningOptions = pioneerReasoningOptions(model);
|
||||
const reasoning = apiReasoningOptions !== undefined || supported(model, "thinking") || existing?.reasoning === true;
|
||||
const reasoningOptions = apiReasoningOptions ?? (reasoning ? existing?.reasoning_options : undefined);
|
||||
const interleaved = reasoning ? (existing?.interleaved ?? { field: "reasoning_content" as const }) : undefined;
|
||||
|
||||
if (baseModel !== undefined) {
|
||||
const limit = {
|
||||
context: model.max_input_tokens,
|
||||
input: existing?.limit?.input,
|
||||
output: model.max_tokens,
|
||||
};
|
||||
return factorBaseModel(baseModel, {
|
||||
cost: existing?.cost,
|
||||
reasoning: apiReasoningOptions !== undefined ? true : undefined,
|
||||
reasoning_options: reasoningOptions,
|
||||
status,
|
||||
interleaved,
|
||||
limit,
|
||||
}, limit, existing?.base_model_omit);
|
||||
}
|
||||
|
||||
const input = [
|
||||
"text",
|
||||
supported(model, "image_input") ? "image" : undefined,
|
||||
supported(model, "pdf_input") ? "pdf" : undefined,
|
||||
].filter((value): value is "text" | "image" | "pdf" => value !== undefined);
|
||||
|
||||
return {
|
||||
name: existing?.name ?? model.display_name,
|
||||
description: existing?.description ?? describeModel({
|
||||
id: model.id,
|
||||
providerId: "pioneer",
|
||||
name: model.display_name,
|
||||
family: existing?.family,
|
||||
reasoning,
|
||||
tool_call: existing?.tool_call ?? true,
|
||||
structured_output: supported(model, "structured_outputs") || undefined,
|
||||
open_weights: existing?.open_weights ?? false,
|
||||
modalities: { input, output: ["text"] },
|
||||
}),
|
||||
family: existing?.family,
|
||||
release_date: existing?.release_date ?? dateFromModel(model),
|
||||
last_updated: existing?.last_updated ?? dateFromModel(model),
|
||||
attachment: input.some((value) => value !== "text"),
|
||||
reasoning,
|
||||
reasoning_options: reasoningOptions,
|
||||
temperature: existing?.temperature ?? true,
|
||||
tool_call: existing?.tool_call ?? true,
|
||||
structured_output: supported(model, "structured_outputs") || undefined,
|
||||
knowledge: existing?.knowledge,
|
||||
open_weights: existing?.open_weights ?? false,
|
||||
status,
|
||||
interleaved,
|
||||
cost: existing?.cost,
|
||||
limit: {
|
||||
context: model.max_input_tokens,
|
||||
input: existing?.limit?.input,
|
||||
output: model.max_tokens,
|
||||
},
|
||||
modalities: { input, output: ["text"] },
|
||||
};
|
||||
}
|
||||
@@ -206,14 +206,26 @@ export function resolveVeniceBaseModel(id: string, name: string) {
|
||||
const alias = BASE_MODEL_ALIASES[id];
|
||||
if (alias !== undefined) return alias;
|
||||
const entries = getMetadataEntries();
|
||||
const normalizedID = normalize(id);
|
||||
const normalizedName = normalize(name);
|
||||
const ranked = [
|
||||
entries.filter((entry) => entry.normalizedFull === normalizedID),
|
||||
entries.filter((entry) => entry.normalizedFilename === normalizedID),
|
||||
entries.filter((entry) => entry.normalizedFilename === normalizedName),
|
||||
];
|
||||
return ranked.find((matches) => matches.length === 1)?.[0]?.id;
|
||||
for (const candidate of veniceBaseModelCandidates(id, name)) {
|
||||
const normalized = normalize(candidate);
|
||||
const ranked = [
|
||||
entries.filter((entry) => entry.normalizedFull === normalized),
|
||||
entries.filter((entry) => entry.normalizedFilename === normalized),
|
||||
];
|
||||
const match = ranked.find((matches) => matches.length === 1)?.[0]?.id;
|
||||
if (match !== undefined) return match;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function veniceBaseModelCandidates(id: string, name: string) {
|
||||
const candidates = [id, name];
|
||||
for (const value of [id, name]) {
|
||||
if (value.toLowerCase().endsWith("-fast")) candidates.push(value.slice(0, -"-fast".length));
|
||||
const withoutFastLabel = value.replace(/\s*\(?\s*fast\s*\)?\s*$/i, "").trim();
|
||||
if (withoutFastLabel !== "" && withoutFastLabel !== value) candidates.push(withoutFastLabel);
|
||||
}
|
||||
return [...new Set(candidates)];
|
||||
}
|
||||
|
||||
function getMetadataEntries() {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user