Compare commits
2749 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 792bf222de | |||
| 03725fc679 | |||
| c82de18830 | |||
| df42971011 | |||
| 45618e6834 | |||
| 0c75fc15da | |||
| c3f4e91ed4 | |||
| 6fdd1cd8c8 | |||
| 3fca1470d4 | |||
| 7348388341 | |||
| e9d587bd37 | |||
| 9898563eaf | |||
| d6f2d746cf | |||
| 566cb47fca | |||
| bc797495de | |||
| 942ae7655f | |||
| e76601cc9b | |||
| 2bf785bb63 | |||
| c410335b0b | |||
| 8823b3a9e3 | |||
| 06e7522d2f | |||
| 6283d8776f | |||
| 1eead6f4ed | |||
| d0dc6891d5 | |||
| f957a50b61 | |||
| 4b014875fd | |||
| 87952a4b23 | |||
| 72290385bd | |||
| aa87b7decf | |||
| 9b29ae2e61 | |||
| 8af698d256 | |||
| 9c3a2fc274 | |||
| b9b4831aa1 | |||
| f1d91a4b28 | |||
| d8381a7ca5 | |||
| 7d0b553389 | |||
| 33d7896605 | |||
| a4fdae1ccd | |||
| eb6599950b | |||
| 3474177f9e | |||
| 7291d335b5 | |||
| e7ad075a8c | |||
| 31e40bc95f | |||
| 0a5389f404 | |||
| 13a3f2cd34 | |||
| 9529fb3de7 | |||
| cc2e96f0ed | |||
| 3dcc9bcc1c | |||
| 3feca23e06 | |||
| 4606f4c9a9 | |||
| 4deddfb90e | |||
| 01442e3d20 | |||
| 1b898e3bb9 | |||
| 5f35c090c1 | |||
| 1b6d7d5eef | |||
| f66e29e995 | |||
| f0492e8773 | |||
| 90abdfdb79 | |||
| 509251683f | |||
| 8eaf0fa58d | |||
| c018094ef7 | |||
| 5905921eb3 | |||
| b6daef9401 | |||
| 7d829b7f8f | |||
| 63a38aea7d | |||
| b394007dd4 | |||
| 62017da0c1 | |||
| e1c2324a29 | |||
| 39860aebd7 | |||
| 172ada0349 | |||
| d2b0493f87 | |||
| 434a80de95 | |||
| d3ff76c1d3 | |||
| f909ea1715 | |||
| fde5ef1419 | |||
| 4e6f75c544 | |||
| 8cd56f6e2b | |||
| 404dd11af5 | |||
| c97375d63f | |||
| b502337c06 | |||
| 325aec0452 | |||
| a66fecc91c | |||
| 3693a52ad4 | |||
| 3160035f0c | |||
| 20d7995919 | |||
| e6b78e1d0e | |||
| 17405da1a5 | |||
| 8c1a80f5c2 | |||
| 5c54cb6238 | |||
| 06dd08ce3a | |||
| daaf9c67c7 | |||
| 475375bdb6 | |||
| 2b963a06fe | |||
| 58201a06c3 | |||
| e6461f24e1 | |||
| 2bb798e030 | |||
| 97fc94a506 | |||
| 601454c238 | |||
| 024266637c | |||
| d38daa49ca | |||
| fa5aaa8e78 | |||
| 2477a59f60 | |||
| c8ef8de934 | |||
| 71db3db82e | |||
| f670a0250b | |||
| 9823b2486b | |||
| fbc95e28a4 | |||
| 527b0fe184 | |||
| 1bfa268c1d | |||
| 3c3f401d71 | |||
| ceab7f093b | |||
| 1629d0e613 | |||
| 247e5f7138 | |||
| 3d15bd19ea | |||
| ecdfc68d73 | |||
| 581196a937 | |||
| be4aff364c | |||
| cd05dd252f | |||
| 7bce4e856d | |||
| c1ec31b651 | |||
| dc199d02a2 | |||
| d7e52a8aa4 | |||
| eeb95cb56b | |||
| 0166f6d997 | |||
| 4ff882e629 | |||
| f23e2cec3a | |||
| 182d8dac52 | |||
| d3e52ae5a7 | |||
| fbc01c01bc | |||
| 352e09c3ef | |||
| 176454ecde | |||
| a761850bf6 | |||
| d3b7a116b5 | |||
| 899c4e31b7 | |||
| ea1f773c09 | |||
| 2d7a8a5bfc | |||
| a9fd4b61d2 | |||
| 4c10fb10da | |||
| eda1040428 | |||
| 5401da01ec | |||
| c919af4bb7 | |||
| 31f80be7c0 | |||
| fac59c22bf | |||
| 8e91264e8e | |||
| 5e49f2a366 | |||
| 6e0d2d9b50 | |||
| 33d9861cee | |||
| 22ec791277 | |||
| e29c5e212f | |||
| 3c8caf5938 | |||
| 9108580844 | |||
| f7d358040f | |||
| 7343c558bf | |||
| 2dc15e9e3e | |||
| a74015731f | |||
| 993b751e57 | |||
| 178d5e5c4d | |||
| 5fcb8509b2 | |||
| f29a908049 | |||
| 376d95d251 | |||
| 0c9f63575c | |||
| d03182b119 | |||
| b3599e6fab | |||
| 7dcb38c785 | |||
| d008d00785 | |||
| fe43b1727f | |||
| bedab81ab8 | |||
| ca0da25618 | |||
| fd6197abc1 | |||
| e3cdc87b36 | |||
| dcf9ab2975 | |||
| f25d350db6 | |||
| 7b7d92d8a4 | |||
| 75568c4d79 | |||
| a64c2b256d | |||
| bb844ab9ca | |||
| 9bf21386f2 | |||
| fe90c89beb | |||
| faa95314e0 | |||
| 15aed038a8 | |||
| 0e756a9ab5 | |||
| 0d3649fe2a | |||
| 7c4393babe | |||
| 81ec280e43 | |||
| b645ea1527 | |||
| 729173a249 | |||
| 33b39db1dd | |||
| 972e8b07f8 | |||
| 1c33d09db4 | |||
| 07ba8d2297 | |||
| 04f42dacca | |||
| 15402b2b68 | |||
| 379c81040c | |||
| 076e6b7a6d | |||
| 5d73be0443 | |||
| 8150634c30 | |||
| cfea58ad4b | |||
| 7db07272d0 | |||
| 0da41199e2 | |||
| f2749e4850 | |||
| 5818117a08 | |||
| 4f5a5ddae2 | |||
| a7a40d4b6a | |||
| 5ebab13429 | |||
| 7873676005 | |||
| 767647b041 | |||
| ec71d51e43 | |||
| 3d377770fc | |||
| c0564a9afa | |||
| 5d2a2074b8 | |||
| 2c5fc7529c | |||
| 64ddfdb59d | |||
| 3828d35408 | |||
| 33317b3d9c | |||
| d1fbb0f644 | |||
| 06f66b1623 | |||
| f27645078c | |||
| b42a6aed71 | |||
| 66b1577ac2 | |||
| 35ad28cd97 | |||
| d3468c3c5b | |||
| 68ba2a6e5c | |||
| 88c88f8959 | |||
| d3a600b878 | |||
| 34c5c2c19b | |||
| 8c97019802 | |||
| 99a34ffb45 | |||
| 26f09f6a60 | |||
| e3d2ab0579 | |||
| 394784f556 | |||
| 38f2509237 | |||
| 4b19304443 | |||
| 9bced61669 | |||
| 86e0258ff2 | |||
| 01363e5e21 | |||
| 57fa4d7c07 | |||
| 7745298340 | |||
| d3b6026794 | |||
| 1eb0f215ad | |||
| 8c18476032 | |||
| a791855b5a | |||
| ef10ebf165 | |||
| 66098dfac4 | |||
| 9ef3965a79 | |||
| d05f01ba33 | |||
| ccf7575885 | |||
| 29ee1c21b9 | |||
| 23ddb04901 | |||
| 6bfe4c3a32 | |||
| a64e9f3057 | |||
| 195a4b32f2 | |||
| cae730d062 | |||
| 3bc6dc1671 | |||
| 5176e956d7 | |||
| 15c1873f74 | |||
| 7b52656170 | |||
| ff1a701887 | |||
| 69ba49bd58 | |||
| fa44c653d5 | |||
| ac7fcfe158 | |||
| 04988b262f | |||
| 691fb97870 | |||
| 4e70a07d80 | |||
| 02dc52e7e6 | |||
| 445564e238 | |||
| c3d5048e84 | |||
| ccc394990e | |||
| b09b28918f | |||
| 424a9cd1b4 | |||
| 34eeabc074 | |||
| 2ecea5618a | |||
| 349131979b | |||
| 561c304567 | |||
| 65d43e7472 | |||
| 04e3998fd8 | |||
| 9b445d4342 | |||
| d6e863584b | |||
| ba12d22b5e | |||
| 39f09aaa2a | |||
| 6305906f8a | |||
| fe72d99809 | |||
| 6f6925f5b3 | |||
| d284951a3a | |||
| d20394610d | |||
| 8d839838dd | |||
| a163b97534 | |||
| c16c5d4685 | |||
| 4f15f73571 | |||
| 37fc5d8fad | |||
| 6218d40c5c | |||
| 308973ff70 | |||
| ec03cfa65a | |||
| 93b0b473db | |||
| e7ab5d5215 | |||
| 29fd5aa151 | |||
| a82f812b72 | |||
| 2425ced352 | |||
| 641e6266d3 | |||
| db68cb6559 | |||
| da34a15354 | |||
| c707841460 | |||
| 68d22ac201 | |||
| 78f50d6612 | |||
| 83529d1561 | |||
| 4abaaaa78c | |||
| 125e69c785 | |||
| cca3855c88 | |||
| fe641e7976 | |||
| 9839ae72dd | |||
| 50fb169a38 | |||
| 716a1e5b2e | |||
| 9e783a316e | |||
| a1fa73f458 | |||
| c95cc7b800 | |||
| f74418bd99 | |||
| 6764794a4f | |||
| c809ad828c | |||
| a8fa130077 | |||
| 2f915de002 | |||
| d7e6d01d10 | |||
| 27d7407f9f | |||
| 44928267aa | |||
| 48a688040b | |||
| 88e8af2bdc | |||
| 9cbf9d15f5 | |||
| 2bbeba0f22 | |||
| 5500871521 | |||
| 71598fa7ba | |||
| 7041cee55c | |||
| 5520268eb4 | |||
| 7c4ee4e68e | |||
| 65efb3410e | |||
| e1b14c5323 | |||
| dd2f9fd121 | |||
| 6fe6842eb4 | |||
| f136dca121 | |||
| 346acb9583 | |||
| 24bc374980 | |||
| 160bf03fb1 | |||
| 9a4c498d6d | |||
| 52765e0022 | |||
| 9a39dcf9b2 | |||
| 44febfc2c9 | |||
| ebfffa1ca7 | |||
| fed4eb76a7 | |||
| 9daf8e545c | |||
| 4ae880a20f | |||
| 465f99c59d | |||
| 0841aef369 | |||
| 674eb9cf71 | |||
| 88147542c2 | |||
| 836873629f | |||
| dc8d3f7e6f | |||
| b1c751597b | |||
| a974dcf82c | |||
| 380b596d14 | |||
| 7a3938e539 | |||
| c4a25e61ef | |||
| 5f40b4b5db | |||
| 5101a9ea79 | |||
| 30f1e75fd5 | |||
| 5ee5ad3e70 | |||
| 2129e63db5 | |||
| 5ae21fa042 | |||
| f66b963253 | |||
| afa1e4f415 | |||
| b7951a7ddd | |||
| 5234c4f076 | |||
| ce0aed5c2f | |||
| e1d5d64012 | |||
| 1e83571461 | |||
| 31d7f3a70b | |||
| 2603544904 | |||
| 3e182ca417 | |||
| baaf6a2116 | |||
| 8346d6aaf6 | |||
| 8e63470e66 | |||
| ad4c44587b | |||
| 04d1f31206 | |||
| e29212ae16 | |||
| 8469f2057e | |||
| dc87a8331d | |||
| 5b3f6007b1 | |||
| a71bb99aad | |||
| 19c03a3844 | |||
| 5acb3ea319 | |||
| 8210ebf55b | |||
| 6a2801c0f3 | |||
| 8c70b10996 | |||
| 0a8cd1ae61 | |||
| 6c8cfb5f5e | |||
| a3414fc55c | |||
| 4e0262cdc5 | |||
| 3c96b9e4a6 | |||
| bbb02dcf8b | |||
| 638f26990c | |||
| 1d124fc929 | |||
| 08c41ae9c8 | |||
| 5509ac0821 | |||
| b6e232ed89 | |||
| 5fe6ebe731 | |||
| 796b2ee795 | |||
| d6a2d17e0a | |||
| a391dcff54 | |||
| bafcd75683 | |||
| 0393cc20fd | |||
| 5acbb4e384 | |||
| 071945fa32 | |||
| ca161dc7c3 | |||
| eba9e31f3b | |||
| 70897b3be2 | |||
| 35248d4cf9 | |||
| 93adad6a1c | |||
| 0ef223fc75 | |||
| fc697b7f8e | |||
| 2a6f94d800 | |||
| 77eb76692b | |||
| c9bd5f1697 | |||
| 94b5abd3d3 | |||
| 961fea555a | |||
| 49c81b155b | |||
| 7a0f1749b7 | |||
| 11ec3cb718 | |||
| 6d2cab9ef1 | |||
| 38d530fbb2 | |||
| 89a7f87416 | |||
| 08f3c771dc | |||
| 8f98c650c3 | |||
| 2aee67ae2f | |||
| 3c0cd62170 | |||
| 3e6809e3de | |||
| 6e6f20381f | |||
| 961cbc89c7 | |||
| d6ffa276ab | |||
| d8465b36ea | |||
| 5803b02ac6 | |||
| 4508b20726 | |||
| 1739b0578b | |||
| b1496b3049 | |||
| 231aa73003 | |||
| 4ce426721e | |||
| 29a1f478f0 | |||
| d4be11a245 | |||
| 7a9e79f26e | |||
| 91ee4af762 | |||
| 9b52b7143e | |||
| a284b59bf4 | |||
| 968af5c536 | |||
| 6d4e298319 | |||
| c7f0b1f858 | |||
| 50cd64adae | |||
| e4486f5dcf | |||
| 85e7f22240 | |||
| 801600baae | |||
| d345b3506e | |||
| 86c39dfd3e | |||
| 7498a53fe9 | |||
| a30f51ccb7 | |||
| f60dce9bda | |||
| 795f5e0226 | |||
| 93982425e6 | |||
| d26592e4bd | |||
| fadbd16d60 | |||
| 3dde9f1067 | |||
| 0e847455e7 | |||
| 9c2fc9d66d | |||
| fb53ad7d35 | |||
| 861458275a | |||
| 5da04090b2 | |||
| 2781dd2b76 | |||
| f86a66fde9 | |||
| 87c4103fa7 | |||
| 180181d157 | |||
| abec199dfc | |||
| 1ace2b0813 | |||
| 0466e45351 | |||
| 8d3a58360d | |||
| c32eb334e0 | |||
| d120fdf30f | |||
| af69924419 | |||
| bd19445c99 | |||
| c08bbecd12 | |||
| a0a2ec27e4 | |||
| 5071a8967f | |||
| 24fa3641eb | |||
| 0644082f4f | |||
| d4049350d5 | |||
| 8dd74775e3 | |||
| 606ffec741 | |||
| 22a0443ec6 | |||
| 6f77c31095 | |||
| c7bd68af7e | |||
| 7e8bd2036e | |||
| 0ba7a97170 | |||
| 8177611995 | |||
| 4ce82b9a13 | |||
| 05f55d2af7 | |||
| 7a6a8767e7 | |||
| 10760bb0da | |||
| d0190e9df0 | |||
| f29cda9f0d | |||
| 5e1dddf150 | |||
| e7b6a156d4 | |||
| 549a7cba5d | |||
| 476b4c49bd | |||
| 1fbcc058dd | |||
| 9752fd1f96 | |||
| 188c2490ae | |||
| f490677b7d | |||
| 150fd13a4a | |||
| fd915d6fbe | |||
| 0724c658b1 | |||
| c777fcb848 | |||
| 7700110cb8 | |||
| da6517ba1c | |||
| 1cde1c8675 | |||
| ed298be418 | |||
| d4d929a4d4 | |||
| 729448077e | |||
| fe8edb96c4 | |||
| e0f4ad4ef0 | |||
| 6e69d727b1 | |||
| 94b76fe501 | |||
| c4869558b2 | |||
| 0a51f5faca | |||
| 202e7f5648 | |||
| 1a1e83568d | |||
| fbce60632c | |||
| f19ee484e9 | |||
| 584876c7e6 | |||
| 118be60a26 | |||
| 093a8ecf78 | |||
| 8ea74a7e65 | |||
| 41baa8fcfc | |||
| 1aef456c4d | |||
| 9596eee8b6 | |||
| 0a5226aecc | |||
| eacb5fb70a | |||
| e4c0f5dafa | |||
| 314062868d | |||
| ccfd58c94e | |||
| 9da6b893e4 | |||
| 21d9f66fa1 | |||
| 3f6fc39407 | |||
| 8342c39909 | |||
| 6f4a4db46e | |||
| 2514e5184f | |||
| 903d4d2642 | |||
| d3212dd707 | |||
| 4e98c819d0 | |||
| 92a3bac8c3 | |||
| dfad4d32cb | |||
| 3b9da5bec8 | |||
| 16056ddbac | |||
| 9fc94e16a6 | |||
| 09c5e71e42 | |||
| e2a7648cb6 | |||
| a71bb2eeba | |||
| 1a58984d3f | |||
| 02a7e65cae | |||
| 17cd180c15 | |||
| 2b9512c782 | |||
| 7ae41a8d3d | |||
| b2b1575c8a | |||
| bb604066ae | |||
| a113c4d153 | |||
| 8e29857b70 | |||
| d747161a10 | |||
| fd96d28cc4 | |||
| 0bf7d83e31 | |||
| 291c944f61 | |||
| 7ccd3fa0db | |||
| b590533805 | |||
| 08db36d2ff | |||
| 4aa9705ed2 | |||
| 1391fb85e0 | |||
| e29b979e4f | |||
| b63ee49388 | |||
| ed2a881822 | |||
| 967956a1f2 | |||
| d66b92e3e3 | |||
| 0fc0aeb9fa | |||
| 937ab7f1d5 | |||
| 2ae1ab8d1e | |||
| 39d5f4d9b8 | |||
| b228827b6d | |||
| a6de2fad66 | |||
| 2006675db1 | |||
| e46764d9b0 | |||
| 27ee6d3826 | |||
| 0477b5ce0c | |||
| 0a6b5e6bbd | |||
| f22aee104c | |||
| 37f641da37 | |||
| bc839fefa7 | |||
| 908b236ecb | |||
| 2379596a19 | |||
| 82aee420a8 | |||
| 2afa12f4eb | |||
| 106476f82a | |||
| 941ce45774 | |||
| cef35890f3 | |||
| 72adad1e01 | |||
| e94a2ce11e | |||
| 0765341954 | |||
| 6138f2d233 | |||
| e20e798af1 | |||
| ca80dbb15e | |||
| beb7ceb031 | |||
| 2bbe3bfb7e | |||
| 7852136b06 | |||
| 7bc898b070 | |||
| f9a3f6d2da | |||
| 4ab1136af4 | |||
| d59211cd0f | |||
| 90843f15bc | |||
| 0381acf584 | |||
| 8fa963fa88 | |||
| 0a88f47a40 | |||
| b6cd2aedfb | |||
| 0b67e8d4aa | |||
| 4c3bb7f342 | |||
| 766af2aa5b | |||
| c4bec67159 | |||
| 6faf36da01 | |||
| b677978872 | |||
| b773a47904 | |||
| b9d0ee13d8 | |||
| 1f9739a8f6 | |||
| a50207f57b | |||
| eaf804b66d | |||
| f41c2fe395 | |||
| c2bd15b59f | |||
| 3bb8d610aa | |||
| 86d7fccb33 | |||
| 1ca5c83441 | |||
| 78696f5e35 | |||
| 41f81c7b49 | |||
| 928c152bca | |||
| 198c2c0187 | |||
| f056021942 | |||
| 6f152c62ab | |||
| 41f6d068b9 | |||
| a322c30f59 | |||
| 542789e8d9 | |||
| 417909337b | |||
| 3fe7ab4a8a | |||
| 7e914b1452 | |||
| e3e377eb36 | |||
| f383bd4c06 | |||
| 340c44b7e7 | |||
| 3a78ae48ce | |||
| 19c1c261c0 | |||
| 859bdd5eb3 | |||
| 8d867e9ad8 | |||
| e3de51a17b | |||
| 2af04645ad | |||
| 0a1161c76c | |||
| 020ea3c7c2 | |||
| ca82ba03dd | |||
| 36a9766893 | |||
| 4569c27dcb | |||
| 5f485767b8 | |||
| 48e2cf17c7 | |||
| ef088cd67f | |||
| bb6866cf51 | |||
| 727059298a | |||
| 33c2087443 | |||
| 8c75b4fc42 | |||
| b169955533 | |||
| ee8c2ebe97 | |||
| 4d968b5b5b | |||
| 701580609d | |||
| c56a70173b | |||
| 42e6665f7e | |||
| 59b8a6b054 | |||
| d8b07e46fe | |||
| 72e14c59ee | |||
| 08b8aea058 | |||
| 955ea42635 | |||
| 45fe490401 | |||
| 3d0e706030 | |||
| e59f9f4a8a | |||
| 5c1ed46472 | |||
| 39532e4797 | |||
| 9b85656b0e | |||
| 4b8317ed24 | |||
| 1063d86638 | |||
| 213680d15c | |||
| 51cf8b43d3 | |||
| a06bb15612 | |||
| 89fcccd697 | |||
| 28a95b2503 | |||
| 59a7c2f8a5 | |||
| 6494e9056b | |||
| f8ef49b8c4 | |||
| 12bfd641d2 | |||
| a06b1d493e | |||
| c763ef8945 | |||
| 1fc631139b | |||
| 338cfd09da | |||
| b4830fbe9a | |||
| 9b5f942c92 | |||
| 790dc87b01 | |||
| 502bc553a3 | |||
| 689e992265 | |||
| c07b4032d2 | |||
| f7c3fe6b41 | |||
| d50b5c2189 | |||
| 5d6ad3885f | |||
| d633b5fedd | |||
| 21b93b3073 | |||
| 7ad4a36d33 | |||
| faf0db274c | |||
| c925183cb6 | |||
| cb19347aab | |||
| bbd4ab17dc | |||
| 60eb28a7fe | |||
| d410b39ead | |||
| 671adf763c | |||
| 4f572fd57f | |||
| 3be139c353 | |||
| d57fc7d3ca | |||
| 898434f914 | |||
| ff9fb4f512 | |||
| 718ae8e156 | |||
| dc2917e828 | |||
| 7d892de5ba | |||
| d95e7792fa | |||
| 796dc7e1c3 | |||
| 2d8b9207c5 | |||
| 998ea25eec | |||
| 696a083dc9 | |||
| 14c2c0ea76 | |||
| 744354b6cb | |||
| a038e8c236 | |||
| ac58aaec14 | |||
| 81304300c3 | |||
| e444a20170 | |||
| ab250abea9 | |||
| 4d0b5d1942 | |||
| 858c3da3cf | |||
| 729fa9cf39 | |||
| 0e7af11395 | |||
| 0a9c79e160 | |||
| 98d9aaceab | |||
| 5589e0f811 | |||
| a767ab0375 | |||
| 8f1de9d193 | |||
| 64ff9445fe | |||
| 7b6be7211e | |||
| 8821c8929a | |||
| 2c7002e668 | |||
| f86f38b2f8 | |||
| a12af3eb6b | |||
| 88a2739864 | |||
| 219196594b | |||
| 7e2169e46a | |||
| 17854e929a | |||
| d69d30fb19 | |||
| 1ddf4e341e | |||
| 5ac6162adc | |||
| 8326bba9f7 | |||
| 598acf90db | |||
| a89e074e7a | |||
| 6fd2e859c1 | |||
| c77017ec66 | |||
| aadd1725ef | |||
| 7643ab5787 | |||
| 7597e1691c | |||
| 362b541a4b | |||
| e768dfeb29 | |||
| 0c5a9bb3c9 | |||
| 98a0944cf4 | |||
| 5d38066662 | |||
| f8ff02ff35 | |||
| 1c6226d159 | |||
| f6348a1249 | |||
| 321e447074 | |||
| 814ad2d85c | |||
| faf3ae09fe | |||
| add87e1d28 | |||
| c74f228642 | |||
| f4955b9ed8 | |||
| 9c33144003 | |||
| 2f1ade02d9 | |||
| b0f56b4714 | |||
| 89c1032d07 | |||
| 5dfc76f405 | |||
| 75f479f1a8 | |||
| fccaf04506 | |||
| 275f068ae2 | |||
| 0c3181f6f4 | |||
| c57e4f6385 | |||
| 6cffa165fd | |||
| 42bfe626df | |||
| f93f721779 | |||
| 61382b27bf | |||
| 0454264f53 | |||
| af4a838509 | |||
| d4dcaa28b9 | |||
| 504228b96a | |||
| d7093921f9 | |||
| 73ea32dcbe | |||
| 16ed31cc3f | |||
| 3bb32c3464 | |||
| 1c0c1ffe7c | |||
| 887c0db7c0 | |||
| e3ddaec403 | |||
| 9740723323 | |||
| b11181f1c2 | |||
| 886a14908f | |||
| d77884c906 | |||
| a020dbb672 | |||
| ae71493b0f | |||
| 99997a8708 | |||
| 81bb6f77d3 | |||
| 058ed3f7b1 | |||
| b6cfb869eb | |||
| 87dcaba3b7 | |||
| 0594c78ce0 | |||
| 22ba4b7ce7 | |||
| e0aaeb0743 | |||
| 309d3628a3 | |||
| 77b6dd661f | |||
| 78568d06fa | |||
| da46af1a79 | |||
| 9b5faa305f | |||
| 3f4d630699 | |||
| 84df20e5d6 | |||
| 4af327c38d | |||
| 6401f6e2bf | |||
| eae4140d06 | |||
| 0f777bf29a | |||
| b4371fc4f5 | |||
| e14499af42 | |||
| 02db1eb868 | |||
| d971f6d381 | |||
| 5b8490caaf | |||
| a17536ca0a | |||
| 6fa6da6d93 | |||
| 593a1b6690 | |||
| 5f1e7545d0 | |||
| 79de9ec9eb | |||
| a5c3c30b1c | |||
| 165a00f2ef | |||
| fded7c917d | |||
| cd15e7a7ef | |||
| 939dc3813f | |||
| 8287b23fa7 | |||
| 2909e8212c | |||
| c267c6ea80 | |||
| f57b3abd4f | |||
| 60111bfdc1 | |||
| 7212d9eab6 | |||
| ff9da01424 | |||
| d04c526635 | |||
| 3e174493b9 | |||
| 285eedc9c0 | |||
| 7d2b7aa3db | |||
| 4260a547b1 | |||
| 9683f634bd | |||
| 9759271d5c | |||
| aa7d1fded0 | |||
| 70a021f122 | |||
| 3de6493dbf | |||
| 21bb25a4e9 | |||
| 21b75486a9 | |||
| 7bcdd34188 | |||
| 96605c24ab | |||
| 6c1885ce51 | |||
| 5a58a73e5d | |||
| c8599d1f9e | |||
| a532ba183c | |||
| bb0c259469 | |||
| 42012f4c57 | |||
| 0ad7bd6013 | |||
| 31fdfe37c1 | |||
| 38627ca1b5 | |||
| 750870cfef | |||
| 0d42429ad1 | |||
| fc4282fe80 | |||
| 78446ab5ba | |||
| 26ba39f18d | |||
| cfb2493d15 | |||
| 0fee8002c8 | |||
| a701d48416 | |||
| 6a62671922 | |||
| a862a612ca | |||
| d55770c7b4 | |||
| 50f58394bd | |||
| 4b066d96a6 | |||
| 7bb1d16eb0 | |||
| 74fc26ea69 | |||
| c85149f6a3 | |||
| e6229e19c6 | |||
| 8666eaffeb | |||
| 5faf9bee8f | |||
| 6217c374e3 | |||
| e8426edf8e | |||
| b79bdbf7eb | |||
| 78c8fe1ac2 | |||
| 5c09f66b8c | |||
| 09d2a8f37f | |||
| b06fce1f41 | |||
| 63b263f490 | |||
| 3984a363d2 | |||
| 9259ec61c6 | |||
| 98260f6acb | |||
| 92cf88ab7b | |||
| 61c6e1efe2 | |||
| b01d057a58 | |||
| 469ae40894 | |||
| c89dec2124 | |||
| 06bdab60f6 | |||
| 959645c983 | |||
| a555f36cb8 | |||
| 450d391ef1 | |||
| 6c718e80c5 | |||
| 75d2ac79b7 | |||
| 89ae1ee0a7 | |||
| 47dd5cac42 | |||
| 559774ef85 | |||
| c460f8eb8d | |||
| 9f29da940b | |||
| c91bb48ac6 | |||
| 88e7433652 | |||
| 131c4ce1f0 | |||
| 5db3ec68d0 | |||
| 72c6b6d40a | |||
| c303859bc5 | |||
| 850844f8db | |||
| cd2e964446 | |||
| d007a41dd8 | |||
| 52dce9c38a | |||
| d9a8197786 | |||
| f28c3474e9 | |||
| 9e1acde4b2 | |||
| bf0650e13a | |||
| 1a84984efe | |||
| b5c69c3faa | |||
| 11758e010f | |||
| d7cf847943 | |||
| c1851cfab0 | |||
| ba838f2ca7 | |||
| da07da2c4d | |||
| 7d17b0730d | |||
| 3ab39c65a7 | |||
| 7a12709f28 | |||
| b75a5f6534 | |||
| 9439c3b7e4 | |||
| d2fad47bf7 | |||
| d747e511b5 | |||
| ee337bd76d | |||
| 8e3a700be1 | |||
| e42d7df665 | |||
| 66328c62cc | |||
| 1317762992 | |||
| eded197269 | |||
| 948ec8110d | |||
| 4497996739 | |||
| 2b32e8cb11 | |||
| febaf8fef1 | |||
| 6ca650d53d | |||
| b2bcac342d | |||
| aae29d4f60 | |||
| 761e859579 | |||
| 3b486955b9 | |||
| d944f72df0 | |||
| ca8a27bb79 | |||
| 4226966d56 | |||
| b5e341e0b5 | |||
| 55f96176e0 | |||
| dd27f6d58e | |||
| 50168cca04 | |||
| 11ba51c686 | |||
| e0a596be96 | |||
| 6aac65431b | |||
| b6159827b1 | |||
| dbc1562762 | |||
| 2f6b896208 | |||
| f8db903526 | |||
| 4adfdb2eff | |||
| 95389ff5a6 | |||
| 3723d6f2bd | |||
| ff5349f410 | |||
| db680b273e | |||
| 1cdc669563 | |||
| a26d25d6a7 | |||
| c40d702300 | |||
| 53d71c61ef | |||
| 63ae17c0f6 | |||
| 8f8e3e28f6 | |||
| 1663d89811 | |||
| ba9275e5de | |||
| a6298265bf | |||
| c68cd4e6e8 | |||
| a2e89bda32 | |||
| 32e01e5974 | |||
| ba30a76db4 | |||
| b1ea28e96d | |||
| d25deb565e | |||
| 9993ba54da | |||
| fa7bce9387 | |||
| 30735edefd | |||
| ae226979ba | |||
| 0a718605d2 | |||
| 1467c442fa | |||
| 27cb77a17f | |||
| 4f690e92d2 | |||
| 26980e3c0a | |||
| a13958f8d6 | |||
| b4c4027d7e | |||
| 5823a5e198 | |||
| 2226b778fe | |||
| 42077ce502 | |||
| d5e2c18d67 | |||
| 12bb311dfe | |||
| eebff243d7 | |||
| 3ffa79f17e | |||
| eca1b7ea94 | |||
| 3330439852 | |||
| a93faba15a | |||
| 0af14b0999 | |||
| 8074ddd03f | |||
| 81cb306e7e | |||
| 60e274d652 | |||
| 2c98095a5f | |||
| 6124e42103 | |||
| 5f8598c1c1 | |||
| 893a206e09 | |||
| 013527f2d1 | |||
| 9c4bf8db3d | |||
| 44229da006 | |||
| 603cf68287 | |||
| 1755c9c60b | |||
| 158878fff1 | |||
| bd45485ab1 | |||
| 1589ce37d1 | |||
| 1880581ee9 | |||
| b7c5b4b355 | |||
| aa76eb7462 | |||
| eb5650ced6 | |||
| a33dac7011 | |||
| 5cf0588ed6 | |||
| 72cedfcaec | |||
| a7d81413dc | |||
| 7915f062ff | |||
| 5d4f4448a5 | |||
| c53d3958b1 | |||
| 3916a2244a | |||
| 8ec5da81c6 | |||
| 57688c236e | |||
| b3570a02e0 | |||
| bf144ec994 | |||
| 7f7e14b3f6 | |||
| 4998e27284 | |||
| fb968b6bd3 | |||
| 3aea72e127 | |||
| 0950795ae4 | |||
| b3b15c2b0a | |||
| e3385f1dc9 | |||
| 521a38379f | |||
| 040dfa968d | |||
| c6ec1503b2 | |||
| 43c9f9d44d | |||
| b3d9ef4e7f | |||
| b510543fd0 | |||
| 2e138dd677 | |||
| 54e9f3324b | |||
| b919bc9a8b | |||
| 3469501ede | |||
| d8f8fa2040 | |||
| 9a372f0c76 | |||
| 8576fb5544 | |||
| 90bb9503cd | |||
| 60f2bd32f5 | |||
| 47a0c4f060 | |||
| 97b079c21f | |||
| 17ccd11f3a | |||
| e4c4077829 | |||
| 6d0db00ec9 | |||
| b4dabb5cb3 | |||
| c6f8ffb3b8 | |||
| bfadb01faa | |||
| 55f22be5ec | |||
| 24246b5a31 | |||
| 34369877e0 | |||
| 902ced5007 | |||
| 105a55f044 | |||
| 37a8159595 | |||
| 88473e5638 | |||
| 741fe43a5f | |||
| 11e35a5bc5 | |||
| 7d6b71463d | |||
| 6c04ef4812 | |||
| 336f0a737e | |||
| 379472016f | |||
| 769428b604 | |||
| 9baa7c16c0 | |||
| 4a901faad0 | |||
| f80a55405f | |||
| 33b44aac8d | |||
| 1e0ca88524 | |||
| fa6db1d7cb | |||
| 5c256f519b | |||
| 2cef83fbb1 | |||
| 69bbdfe868 | |||
| bf1f9b0683 | |||
| ccf43aa497 | |||
| 27ff838906 | |||
| 4cf84c8eee | |||
| e3ed68cd6d | |||
| 5a74b49b24 | |||
| b1b829ae26 | |||
| b1e00b48d6 | |||
| a63661272f | |||
| 3e6f0d3e0b | |||
| 161dff1094 | |||
| d9e1559b86 | |||
| 18e0d92240 | |||
| 631524c175 | |||
| 67ad341290 | |||
| caff7f8f28 | |||
| e07f70d18b | |||
| 6a43b63282 | |||
| fca3f5edec | |||
| 56bc3c7b59 | |||
| 3bd2c79986 | |||
| 9aec2a4857 | |||
| bf673b7cea | |||
| deb207b61a | |||
| 45204d1b3a | |||
| 72bf201d84 | |||
| fc32bfa437 | |||
| 46f1cec36b | |||
| aac09e443e | |||
| c88569b374 | |||
| 01f9da9b69 | |||
| e9de82a376 | |||
| 159b9e428c | |||
| 75fb854ff3 | |||
| f39906047d | |||
| 60e687c808 | |||
| f722b0b488 | |||
| 96e3d1491c | |||
| 0166a981a9 | |||
| 9736dc860f | |||
| 0138c6704c | |||
| 6489519d44 | |||
| 90631663ed | |||
| db5a62fcc5 | |||
| 82ea50d5eb | |||
| 4687d63d55 | |||
| 775e693ed9 | |||
| 496e6be72c | |||
| efa5088212 | |||
| db1e3a81fd | |||
| 495d81c6c2 | |||
| f9f2e5948f | |||
| 3d250bece4 | |||
| 8a8565b125 | |||
| 9d25799e58 | |||
| c9a4d3920d | |||
| e670696e57 | |||
| a5236ce1bc | |||
| a09adf5870 | |||
| f4da3021e2 | |||
| c376b67ef2 | |||
| 722a39cfe4 | |||
| f4c50833bd | |||
| 1eee8c97b1 | |||
| 1e6e3aeff8 | |||
| 4a89992e71 | |||
| ffd1192a47 | |||
| 771fb355f0 | |||
| 087416edf3 | |||
| 1767823697 | |||
| a271497e70 | |||
| c421ab8996 | |||
| ec69a44ee5 | |||
| 932d50bcfe | |||
| e8b9466eb4 | |||
| 776f0ab157 | |||
| aa07808ebb | |||
| dc2a77a443 | |||
| cb2d558822 | |||
| 43b1c2b248 | |||
| 078355410a | |||
| 8f47082161 | |||
| d1fdf677d1 | |||
| 653ab7c9c4 | |||
| 3b744f1356 | |||
| e1e8d50eee | |||
| 124b07677e | |||
| 3e8eb7a804 | |||
| c5fe39ff00 | |||
| 7fec3155b4 | |||
| 07d1a865f6 | |||
| f3136a332a | |||
| e4209ec103 | |||
| be229e3e63 | |||
| fbaa0ee954 | |||
| 5e3109c6e2 | |||
| fe4fe89ea7 | |||
| 7fd1417797 | |||
| 429bfde7ec | |||
| 80a134a072 | |||
| b02fe0b769 | |||
| 614657f876 | |||
| b180ecb6fb | |||
| b575db7255 | |||
| 177e8a8851 | |||
| 732a9b1029 | |||
| d3f7168fd2 | |||
| cc01f5be65 | |||
| 8df2a2ebcd | |||
| 3b765ab4e8 | |||
| 1d9e440467 | |||
| 7a6b5be4ab | |||
| e09d928407 | |||
| 73ebddfe38 | |||
| 17b91fedc5 | |||
| 4e8a0e640c | |||
| 29fea93659 | |||
| 3261be0ee9 | |||
| f51325b511 | |||
| 371e166cae | |||
| 2087f18020 | |||
| aaab7717ac | |||
| ab37e35f0e | |||
| b52b0e63ce | |||
| 03d6b2046d | |||
| 640c606c40 | |||
| 46ed084c67 | |||
| 515dc5d5c4 | |||
| b9f3555078 | |||
| 1a106cb297 | |||
| 1475eaf406 | |||
| 8b9b47594c | |||
| b0a461bcf6 | |||
| 4a28dc1bb5 | |||
| c75840e3ab | |||
| 793d20da81 | |||
| 6a45082f76 | |||
| 7039bacb8d | |||
| 4ab906d3a1 | |||
| 50828469e7 | |||
| f7a88d4df3 | |||
| b16e967dae | |||
| b34673f98b | |||
| 69f1b02555 | |||
| f973b7013a | |||
| 62f995ea42 | |||
| 64e5365107 | |||
| 3b6d8c56fa | |||
| 913f7f97e6 | |||
| 314dad10f3 | |||
| 329c6e177c | |||
| 4381f54b78 | |||
| 5e3f25ebbb | |||
| a49dd9c451 | |||
| 7073fdfc0d | |||
| 0ae59259b2 | |||
| 4c150154ac | |||
| f4d6ae1349 | |||
| 4db197e1b5 | |||
| 53a44a5180 | |||
| fce47f5144 | |||
| 7aa4986eac | |||
| 2a09c5b059 | |||
| cfebebc6c1 | |||
| ae28355196 | |||
| 11ba51b701 | |||
| 27dda16e44 | |||
| 7053f47668 | |||
| 20efc4ded2 | |||
| 4e96b02e8b | |||
| 8110bcef06 | |||
| 826b4ab184 | |||
| 5f8d54561e | |||
| e14f01349c | |||
| b069585b7e | |||
| a3058256ad | |||
| 021b7e7283 | |||
| 149027e027 | |||
| 6f49f55a85 | |||
| 56603849aa | |||
| b100a5919a | |||
| aee09b5713 | |||
| a6eb92ab7c | |||
| cf9b48da2b | |||
| ddca84dbac | |||
| c20e1e211f | |||
| 1b3a64c762 | |||
| acbddd0d25 | |||
| 96136042e5 | |||
| 85aa1750fa | |||
| 1cafe7c3a5 | |||
| 99d8d4ad54 | |||
| 57c50e666a | |||
| b2c17e9e4d | |||
| ec3456d9c5 | |||
| 8705bdef71 | |||
| 000233bcd3 | |||
| 0dd25e3ce1 | |||
| 7bb9881d24 | |||
| 7ea7273a09 | |||
| 090443d522 | |||
| c4b2bc3ccd | |||
| e8d752f6c5 | |||
| d2951194d9 | |||
| 0d44485e9b | |||
| f225ef5460 | |||
| 73d129936a | |||
| 06ceeac04e | |||
| c4d6df114f | |||
| a19e5403ea | |||
| d965f5cbbf | |||
| b6c420dfdf | |||
| 1eaa823791 | |||
| cb37684f6e | |||
| f308e4e6e1 | |||
| 99e06f4109 | |||
| cebc23ec0a | |||
| 6d5be82921 | |||
| 4f238868d6 | |||
| 9bfd49816b | |||
| b7e8eb581d | |||
| 9f76ff616c | |||
| 1e22093a0f | |||
| fe56c31b48 | |||
| 42b1619624 | |||
| 5c9652922b | |||
| a0cf33e1fe | |||
| 426ea03fc4 | |||
| fabebea1e2 | |||
| 114f0cf715 | |||
| 1700078899 | |||
| 3379c39d28 | |||
| dfabfb4927 | |||
| fea447c8d1 | |||
| 25aad25158 | |||
| ee1a07f28a | |||
| a07c93c882 | |||
| ebf2bbda3c | |||
| c22b082650 | |||
| 9270cd3be6 | |||
| da2c83c125 | |||
| ea5b71e935 | |||
| e02862972a | |||
| 20dabb8d80 | |||
| 2acc010e6d | |||
| f9c5144f6f | |||
| 7005c9be99 | |||
| 77aeb7928e | |||
| eb7e7a6cec | |||
| 7a6217ca99 | |||
| c4ff051c57 | |||
| 43af92a9f5 | |||
| 2c053d547f | |||
| 1ecce09646 | |||
| a5c2dcce1e | |||
| fd94c9130a | |||
| 467bd99ef3 | |||
| f5aade7bde | |||
| e470253b71 | |||
| 42fc5088f5 | |||
| 2ea16c9538 | |||
| e069872d73 | |||
| f89608c82e | |||
| e4705abe54 | |||
| 9eb5a19d3b | |||
| fb29ceae8a | |||
| 37c4046c90 | |||
| 32658a1489 | |||
| 1a43969562 | |||
| 5d40ba135d | |||
| e7caf2ecd4 | |||
| 713104fbb7 | |||
| c381d6998b | |||
| 08c091a682 | |||
| d2cfbf3680 | |||
| 6002aec7ff | |||
| 9e30f42124 | |||
| a5437130a5 | |||
| c222bc97af | |||
| 8444be6e74 | |||
| ae20a69f23 | |||
| 3da93fadf5 | |||
| 3e0b7b4737 | |||
| 8fa77f5026 | |||
| d2cff84373 | |||
| a081d104e9 | |||
| 78f75b1c01 | |||
| a34aa7c951 | |||
| e3b1d57c13 | |||
| fd6b4193a8 | |||
| 024e6ffb45 | |||
| 7947b0c7a7 | |||
| 1121d459d7 | |||
| 173e3f8ed1 | |||
| ca7d3b25a3 | |||
| e5e7abe935 | |||
| beca558768 | |||
| 01dadf9d91 | |||
| ff215f6674 | |||
| 333696f7d7 | |||
| d11b21a5d6 | |||
| 2d62fb1779 | |||
| ca12f96f9d | |||
| df4e6f921a | |||
| eeb5237b21 | |||
| 1f56ccd7cd | |||
| a466839f37 | |||
| 7831e9e86d | |||
| 8a37ba5dfe | |||
| 6255bd19e8 | |||
| 88a2665962 | |||
| 2277dc457f | |||
| ac9d399423 | |||
| 2d40de2ece | |||
| 78fbe45db1 | |||
| e523acaba9 | |||
| 1219671885 | |||
| db9b4eed8f | |||
| 7bb4c9f7d6 | |||
| 2c55e47724 | |||
| bf8e2a2be2 | |||
| 9707996d28 | |||
| 2e32651c4e | |||
| b243b97cd0 | |||
| 182bbd9ff2 | |||
| ce03b65ad0 | |||
| 7e2a1b6916 | |||
| 53e3dd8598 | |||
| 8e10627a06 | |||
| 5b0217512d | |||
| 7c03dfeae3 | |||
| 0cfb01cb80 | |||
| 6d15eaa6a1 | |||
| 798308cf1b | |||
| 85cf570bb5 | |||
| 375382b9b7 | |||
| f554370d79 | |||
| eafe9acae0 | |||
| 5d2576f7f7 | |||
| fc5dc2c782 | |||
| d09d123d66 | |||
| 2eaa7753d2 | |||
| 58a27a42db | |||
| 6d2aed3476 | |||
| af0bbbc49d | |||
| 6fa31c41ef | |||
| 18c6fc3047 | |||
| e9c3e52e26 | |||
| ea2382132c | |||
| 189aa9d7c6 | |||
| c6f3958975 | |||
| 4cbed275e4 | |||
| 44611b9636 | |||
| 117358d4dc | |||
| 50e5a7ef72 | |||
| c342da0acb | |||
| e3fc80c383 | |||
| 3f0f985c5e | |||
| 5f1aeb52bd | |||
| a281f8630c | |||
| c02a9f5f96 | |||
| d3b5d356c3 | |||
| 5d39b75d38 | |||
| c66423ef93 | |||
| 929ed6ce62 | |||
| 7a48ff8135 | |||
| 7c26a0cebc | |||
| 8ab6066c95 | |||
| d1d8558dd8 | |||
| 5ffcca8be9 | |||
| de89b8b07a | |||
| cef0591522 | |||
| 4152b737a7 | |||
| 3e3444134e | |||
| 1071cb5d49 | |||
| b02cd51fcf | |||
| 988ff95b87 | |||
| ac7b813ebc | |||
| 0e8728c8cc | |||
| 87a19b16a6 | |||
| 36efffffd9 | |||
| f6ca4d8fb9 | |||
| 0b979a682a | |||
| ee82d50584 | |||
| 027d0bb69e | |||
| e6e91e3a9b | |||
| 465ad3205f | |||
| 0d1930758a | |||
| a8201ed70a | |||
| a722392d34 | |||
| 9f8dd0cee4 | |||
| d9b4d011d9 | |||
| ac48be2ca2 | |||
| 67c333f02b | |||
| 7cae29ee61 | |||
| 2e2566c747 | |||
| 351f4d93de | |||
| 2044a0a293 | |||
| 8c2c3834a1 | |||
| c489343ab7 | |||
| d11e7f58dc | |||
| 2b4fbe67e5 | |||
| 03caa7e551 | |||
| 6009732c8f | |||
| 7f463b4752 | |||
| 85f5f00f34 | |||
| 6939664396 | |||
| 1d5640e77c | |||
| 586680c206 | |||
| 62fa3f94d2 | |||
| 5a3e9a5620 | |||
| a6cfc63ee8 | |||
| d5d0417cd1 | |||
| ccb82c222a | |||
| 4a60fbeb4e | |||
| 01540f4e14 | |||
| 85f5c3acef | |||
| b85f506c78 | |||
| f14603e00e | |||
| 3422e09743 | |||
| 8091421fc7 | |||
| 0b1e1226ce | |||
| 6fef96b1ca | |||
| c46d05e824 | |||
| 5878b78d65 | |||
| 1fcaf7b6b5 | |||
| 035c20f506 | |||
| 86db2a1652 | |||
| 4c0be30ae4 | |||
| 52487e1e8a | |||
| 6f6df8c40a | |||
| 3b09d2e1fb | |||
| 24d0ab6e58 | |||
| c29a30bcdf | |||
| 9fe4982e7d | |||
| 6802ce8100 | |||
| f9cbfe67cc | |||
| a385fb1a3a | |||
| a4ac9daee5 | |||
| ffc83193ca | |||
| 6067008226 | |||
| f75366ca59 | |||
| 4a451aba7f | |||
| 5d4f3a6820 | |||
| 413b5900e8 | |||
| 52ff038b55 | |||
| bdac20724e | |||
| 6fdcff1712 | |||
| 7d99872127 | |||
| 45acdbe8bc | |||
| eb4113bbf7 | |||
| 18857af925 | |||
| 019af2710e | |||
| 528ae75ffc | |||
| a0de767e78 | |||
| f598b1ab25 | |||
| 15dfbdac3e | |||
| 6f27e67224 | |||
| e557b85fc2 | |||
| 679c165b12 | |||
| 5c7af27d02 | |||
| 912679972e | |||
| b8b066014f | |||
| a6818b3843 | |||
| 0c988f0fe5 | |||
| 92ea3da07a | |||
| da1a0bcd13 | |||
| 1124adbf0a | |||
| 41f493ec26 | |||
| f55fd15348 | |||
| 323411aa3d | |||
| 1473c33397 | |||
| e314c89a18 | |||
| a12f7d3de9 | |||
| ad01a27c85 | |||
| 19d5953543 | |||
| ded07912e4 | |||
| 11ad8cf1c7 | |||
| a63e6ab6e0 | |||
| dec8dc9b3f | |||
| 9fd7efe800 | |||
| ea209090c2 | |||
| 20c71f4ce2 | |||
| 98694dede1 | |||
| 2b961e4468 | |||
| 5d95219806 | |||
| bd438076d0 | |||
| c5349bdcdb | |||
| 4af30c532a | |||
| 67e2339336 | |||
| 6caf3896d0 | |||
| c42ab3b039 | |||
| d3d2d859a9 | |||
| 3dbc20c7ab | |||
| fa83eddb66 | |||
| eed7c4b487 | |||
| 065e7fc2a4 | |||
| bd796fafda | |||
| da9918dbc0 | |||
| 56aa56ecfc | |||
| 49b5c7475f | |||
| d1caab1165 | |||
| 3748d7b0fa | |||
| 4970c94600 | |||
| ab43664040 | |||
| 64ef89384c | |||
| 57410a206a | |||
| 0aba1466c1 | |||
| ba47f7e782 | |||
| af85ba84ea | |||
| 8a59e93c83 | |||
| 4d484f97eb | |||
| 4291550e57 | |||
| ee29fd33c8 | |||
| 66ef77dba3 | |||
| 9da380b32e | |||
| 1d66f72f99 | |||
| a930b2725c | |||
| c8a5545c97 | |||
| 656d113143 | |||
| d076b25f14 | |||
| 591d24cee3 | |||
| 8ad201a6a1 | |||
| 9f65100344 | |||
| 0a41a6927b | |||
| b1df4c0959 | |||
| 53ed0e946c | |||
| 385679fd1f | |||
| 4fdc18070a | |||
| 04b4fa9e79 | |||
| da84365ed1 | |||
| e9fe601a4e | |||
| 3a788ae9ad | |||
| 55e581abd5 | |||
| 7d66242967 | |||
| 070a3280d9 | |||
| cd48987f94 | |||
| 7bca5bea11 | |||
| 4a76f6d167 | |||
| a8f26dba8a | |||
| d9f90176e4 | |||
| b9b396039b | |||
| 4915da3378 | |||
| 0d404d5b7b | |||
| b875c19197 | |||
| 934f36b3f4 | |||
| 98fc6f0958 | |||
| 6d922badde | |||
| cc8142ceef | |||
| f4597a561f | |||
| 1e2b5e757e | |||
| 8a1efe1c10 | |||
| 1bbe06b189 | |||
| 4deffa8944 | |||
| e3865b185b | |||
| bbab73e43b | |||
| 17460ccce9 | |||
| 9b1821666a | |||
| cadb58e1f2 | |||
| e3bdb4e4fd | |||
| bd4508f37b | |||
| 1b587ff52c | |||
| ff26a32b0f | |||
| 6b9749829e | |||
| ee10c0f5a6 | |||
| a58041ce0e | |||
| 4ec88fb78e | |||
| 6772f4b9fc | |||
| bbcb92a0eb | |||
| 93188d006e | |||
| df4d9d6e76 | |||
| 7425925ef5 | |||
| a19e91e615 | |||
| ffc813e0df | |||
| b99d1bc982 | |||
| 10480737aa | |||
| bbd11f3d43 | |||
| 22be038cb3 | |||
| a2e864c86f | |||
| 3e6aa6752d | |||
| c06a5f0c3f | |||
| e6bf853b36 | |||
| 5128e06105 | |||
| 2f6dec2bc3 | |||
| 26e78b3c46 | |||
| 6aeaf9fce1 | |||
| 42c1d71d34 | |||
| 4470a7bab3 | |||
| 49d68ea2db | |||
| 79c54c4224 | |||
| 1c2d4674aa | |||
| f24ac4cfca | |||
| e813c30bfb | |||
| bbe47c119a | |||
| 68fc7a152b | |||
| 856d298876 | |||
| 78cba12cca | |||
| 6454955e40 | |||
| b4ba8a2fd8 | |||
| 50461628c0 | |||
| 85f63bd3f8 | |||
| 9a004ede00 | |||
| 2a22bbe68e | |||
| 9de824c73d | |||
| e30f8040dd | |||
| 0c03e3aa43 | |||
| cbd597fd3e | |||
| 613b9c1be8 | |||
| eba8a9730b | |||
| 5daec1d50f | |||
| 15d9635ce4 | |||
| dc88cd8299 | |||
| 0f85068066 | |||
| d78b24c13a | |||
| 646970b071 | |||
| e3743f0f90 | |||
| db3cfd4023 | |||
| 1e6f28d5a5 | |||
| f01bd3dc9c | |||
| 35aa838a20 | |||
| c66ec61970 | |||
| 27d4dc4491 | |||
| e7cec38ed0 | |||
| 3d6041a815 | |||
| 3ddd003ea8 | |||
| eca0f14195 | |||
| 5bf20faf51 | |||
| 3a3efd65cf | |||
| 191bf11a51 | |||
| 2b18e6aa98 | |||
| 5b21d3e48d | |||
| 1fca9ba633 | |||
| c973f5bc02 | |||
| c27cd817e4 | |||
| 2024b4b77c | |||
| 4162d897c2 | |||
| 1022889bfa | |||
| 9690429e4c | |||
| 7076cf47ae | |||
| 10d5356dac | |||
| ac6f1bff29 | |||
| a32e82c923 | |||
| 31c7d29855 | |||
| 7a3f040cf5 | |||
| 09b3abf7ff | |||
| 57dbf97205 | |||
| 28f83e40c9 | |||
| d6a53c8712 | |||
| 8d998b719b | |||
| 10e033a28e | |||
| b82f92b7c4 | |||
| efad43fbc2 | |||
| 5bf1d97325 | |||
| 4d57c97b1f | |||
| a3325db6ca | |||
| 66f3798e81 | |||
| 6c31bd08b4 | |||
| a32ebb1127 | |||
| aff9d3a26f | |||
| 20b19ec44d | |||
| 59ffa1318f | |||
| 9e07ccb728 | |||
| 19b875fa6a | |||
| afba33f1fd | |||
| b480109e55 | |||
| f286db67c8 | |||
| e543e72803 | |||
| 5df1e293cc | |||
| e1a2306d12 | |||
| e10d5ea672 | |||
| 7178a77996 | |||
| ecb7dd0cdd | |||
| 59fb942b9d | |||
| cec5a6963c | |||
| 54bc443d72 | |||
| ca42938ec7 | |||
| cf1cdb624e | |||
| f0a1e66372 | |||
| 254e17fe39 | |||
| 35bd218c0e | |||
| b8c28ed06a | |||
| 926d086c52 | |||
| a5e22e7087 | |||
| 265ffc2d8b | |||
| 5c1ce9c015 | |||
| a4f666d17a | |||
| 3f411e5933 | |||
| 1b478d6964 | |||
| e51a3af75a | |||
| 418e99f87b | |||
| 01c37c2bda | |||
| fe9da1a0dd | |||
| dfb98553ee | |||
| 71f35dd578 | |||
| c893b682a3 | |||
| 5fabef3235 | |||
| b7a3f12a9e | |||
| 79a4c41a2d | |||
| d48c92a0ce | |||
| 1a85e7da1a | |||
| 282dee5410 | |||
| db0c5c202f | |||
| e5642a1b0f | |||
| 4ec1138f97 | |||
| f124bd0e85 | |||
| 497aa7056d | |||
| 8de5419820 | |||
| 54e36a83c7 | |||
| 1be4d87a8a | |||
| 78c5e4a52e | |||
| 88ef40f2c6 | |||
| 9885b44af6 | |||
| 6726107954 | |||
| 23527c7da0 | |||
| e704dcb4ff | |||
| 49505b7729 | |||
| 7ed9593b46 | |||
| 8914b0fcf5 | |||
| a818197fca | |||
| a2e4538e38 | |||
| 61135326c7 | |||
| e05359452f | |||
| 807ac13297 | |||
| 8939aa3e5b | |||
| de0dfa6c50 | |||
| 479fc54d84 | |||
| 00f0f9fc2c | |||
| eb05db4971 | |||
| 45c463e11e | |||
| 5a12fedf11 | |||
| e7773692c5 | |||
| bcb0abe566 | |||
| a434bbeea2 | |||
| fc2be2dd1a | |||
| f883cfc04c | |||
| 5c80655ebb | |||
| a1743bbd12 | |||
| ece11f1fc7 | |||
| f06a230f3d | |||
| d486a211ff | |||
| 329e4bbfe5 | |||
| 81f1312d01 | |||
| 991a9837d3 | |||
| e24f8446a7 | |||
| eb03b60865 | |||
| 05e2f2fb8d | |||
| 5a59ace620 | |||
| fd1ab3d0be | |||
| 115734b9d7 | |||
| 5f09e6ac77 | |||
| 1c1dc8743e | |||
| b7021e8be4 | |||
| 486f731162 | |||
| e7ba31e3e1 | |||
| fa9cd05907 | |||
| 65f04ee456 | |||
| 11c60a4e83 | |||
| f70eb46531 | |||
| 94301f1a8b | |||
| bc3dbee852 | |||
| acd07acce3 | |||
| d4bf3cbfae | |||
| 5c83b7d0c7 | |||
| 49128cb40e | |||
| fe4255d100 | |||
| b8537a52aa | |||
| 9fbc9f0be8 | |||
| efd32278c7 | |||
| fa7345c3e6 | |||
| 526fc0e556 | |||
| 921c9e2dd8 | |||
| 8717b868a0 | |||
| da2a668cbd | |||
| 1c400a7546 | |||
| 4dc84304ce | |||
| 0f08c573d6 | |||
| 9255919595 | |||
| 513bd4d6bc | |||
| e327a2371e | |||
| b523c7254d | |||
| 90e6f5bb44 | |||
| c674cef1c8 | |||
| 25f1b08cc5 | |||
| a8403b6eb5 | |||
| efdaed218c | |||
| 822d072122 | |||
| 728ba92d11 | |||
| 2c9b99106d | |||
| 07b090a3c1 | |||
| b765e5aaba | |||
| 72da90d58c | |||
| 894a7e6e9e | |||
| 452e090a55 | |||
| 4d1db24144 | |||
| 430439c272 | |||
| fb25faea9f | |||
| d388c479f4 | |||
| 60fad6b15b | |||
| fcc4d86b0d | |||
| 3187182b60 | |||
| d6467e7fc6 | |||
| 26238aaee1 | |||
| bef1878c44 | |||
| f6c5ba4935 | |||
| da158b062f | |||
| 2c5cbc09e7 | |||
| 63f2df7e4e | |||
| 9fa29b085f | |||
| ff369386a9 | |||
| 22d10c3a6e | |||
| 02c6d94a31 | |||
| 33e38f1094 | |||
| b6b726db04 | |||
| f75df8f6b5 | |||
| b318645646 | |||
| 131ab23cc4 | |||
| 77bff3d4b7 | |||
| e847764971 | |||
| b74de09d8e | |||
| f39d9ca830 | |||
| e66c4b2d34 | |||
| 4d3d175a1e | |||
| 04aa385ada | |||
| f60443543f | |||
| 5327cf33ae | |||
| 7d6c2dfd3e | |||
| 972ba6b977 | |||
| 95f4fb7c45 | |||
| a4e40ed8bc | |||
| 6412613162 | |||
| 6b57d5286f | |||
| 06b6ecfcac | |||
| 52386f616f | |||
| c9552e2dcf | |||
| db641bfaf0 | |||
| d8d38caf37 | |||
| b7d0a720d9 | |||
| c8d7b546cd | |||
| e2c61ed2a3 | |||
| bdba272748 | |||
| a87ad1fbf9 | |||
| 63ad154f61 | |||
| 58429c79bf | |||
| de4206c899 | |||
| c68640f077 | |||
| bbe08204f8 | |||
| 062617cfd5 | |||
| d8bfeace8b | |||
| f9fc094193 | |||
| 0b35f56c82 | |||
| 09ee1b663f | |||
| 18218d5a77 | |||
| 86a89815a6 | |||
| 6047525c92 | |||
| 092af9249c | |||
| bd01b94608 | |||
| 8a556ff795 | |||
| 7cbf065081 | |||
| 3650315e7d | |||
| cbbe4af865 | |||
| 2b66fc266d | |||
| 62b8d68999 | |||
| 7ea4e1726f | |||
| c7f6740807 | |||
| 96e4ba5751 | |||
| 93bf2b0613 | |||
| 0aefd15823 | |||
| 6719022359 | |||
| ebd5fe8b31 | |||
| 71fbc5aebf | |||
| 0a27dacc46 | |||
| 11122a20d8 | |||
| 0ab0f22a17 | |||
| 15ab99712e | |||
| 2154880fba | |||
| 47510448d1 | |||
| e12c155630 | |||
| 8fcdfd0215 | |||
| 9e3f13d0dc | |||
| fda4e85656 | |||
| 029cfd22d7 | |||
| 9d467d62f6 | |||
| 26f685d548 | |||
| 95bc015da0 | |||
| e87bc93b41 | |||
| 290a4a894b | |||
| 8c67f6c05d | |||
| 5f1b9fa7d3 | |||
| 61ccce00f5 | |||
| 3f47c9f65a | |||
| 2b562a0e8e | |||
| 8a4d593e6c | |||
| d3def826ab | |||
| 76480c5c0a | |||
| d68c2e6d5c | |||
| cac41a6349 | |||
| 921244be26 | |||
| 3f5b945044 | |||
| 43d2d1dd22 | |||
| 68d40a1692 | |||
| 5988e1d0e1 | |||
| 703e50353c | |||
| db30d491d9 | |||
| c42f2ce398 | |||
| f9ddbca39b | |||
| 0c6668e56b | |||
| 1e3c39b91d | |||
| 0b1500f88c | |||
| 50d9e22b95 | |||
| f225456039 | |||
| fef4683e8d | |||
| 89312355fd | |||
| 860655a181 | |||
| e52a68063b | |||
| e65f7cea63 | |||
| d62da36a8c | |||
| 66cd6c637c | |||
| 752257fa57 | |||
| 6f6a95b0b9 | |||
| fc6251ae31 | |||
| d6b585cde8 | |||
| a493f144ec | |||
| 03a05b1b1c | |||
| a807490ee5 | |||
| 45da7f314d | |||
| e91209e3d6 | |||
| 9b511fb5ef | |||
| dc5e261637 | |||
| 1a8e068165 | |||
| f48da08f58 | |||
| 1a9e7f4798 | |||
| 5255773694 | |||
| 526188a5ab | |||
| 939c59f063 | |||
| 4162ae60da | |||
| d36fab2afc | |||
| 0eb06b7bfb | |||
| 6320e71132 | |||
| fbc9524ddf | |||
| 7fe4bc609b | |||
| 3361bc1bce | |||
| aa88cc999c | |||
| 7c2adcd28b | |||
| 2c2a89c010 | |||
| 550700f2fd | |||
| c9fbc1961c | |||
| 29325112ad | |||
| 987ccbed97 | |||
| 1394886236 | |||
| 8dd8e9fe79 | |||
| f08c190e25 | |||
| 9c34cf97ac | |||
| 67bf57b3e9 | |||
| d4170dfc97 | |||
| 6a208d0d52 | |||
| 80a6c74f67 | |||
| 4bb6f39894 | |||
| 7e6037c940 | |||
| 9831a4558e | |||
| 07543513c0 | |||
| 3e14ab2b0e | |||
| a7d99ffd8e | |||
| 9bc5f2aa34 | |||
| 529168f43f | |||
| ed9fd5ba66 | |||
| b1e0d66db1 | |||
| 6305b5d229 | |||
| cd7d1b8281 | |||
| c1c9844396 | |||
| e9e930018f | |||
| 3a7e577fff | |||
| 5644dbed4d | |||
| 106a533001 | |||
| a9f03f53d3 | |||
| 5a35cc47a0 | |||
| 19e55dbdec | |||
| 19f8d7f6f4 | |||
| 803301912c | |||
| f4c782bcf9 | |||
| 16d0be15bc | |||
| a654800742 | |||
| 5dc5b44d5c | |||
| 5daab85063 | |||
| 07debd37a0 | |||
| c0ee1388c0 | |||
| ea54509e37 | |||
| caa769e0d8 | |||
| 122e38adc3 | |||
| 9e298b81cd | |||
| 02bec8bb45 | |||
| e3b88068da | |||
| 24340298ba | |||
| 31abd13190 | |||
| 9398b66381 | |||
| 29e4a1d1c5 | |||
| e782c13782 | |||
| f9ef35f8e9 | |||
| 536e5e9ace | |||
| d7fed46efb | |||
| 273ca33f18 | |||
| 3877a8ecb2 | |||
| 909adb1bfe | |||
| 304af83192 | |||
| 9cb06a43f3 | |||
| 7d4edbbc85 | |||
| 23c3c1f9a9 | |||
| e0d0db715b | |||
| 267c84f72b | |||
| 2ffc506fa8 | |||
| 54f41524cd | |||
| f081fcad60 | |||
| 8f02d5ca0c | |||
| 98d16c56d0 | |||
| bbe750699c | |||
| 51b6921c55 | |||
| 32ea08a271 | |||
| e9729f9e34 | |||
| 95f347acb1 | |||
| 95f4ede84d | |||
| 1da95574e9 | |||
| dd8b3a6553 | |||
| 55e6d2365d | |||
| 50d47fd920 | |||
| 5d776a634f | |||
| b96a71b4e3 | |||
| c820b5376f | |||
| c0fa75294b | |||
| bc4ae52ba5 | |||
| 6030c60edd | |||
| dab2a7a0a7 | |||
| 2513766ba3 | |||
| 24759f9fd2 | |||
| e5e6deb5a2 | |||
| 9df613e1d8 | |||
| 16aa7b4269 | |||
| edfe88346f | |||
| c12ce9817b | |||
| beb3c08e11 | |||
| 3be9d8ded2 | |||
| 6c21d3287e | |||
| 17879c190c | |||
| b457193436 | |||
| 5afb63f5ca | |||
| 9343f93831 | |||
| 9f8755d475 | |||
| 2dfeb2a596 | |||
| 91c901eaa6 | |||
| 00877c399f | |||
| 1f4fca1832 | |||
| 1be7f13552 | |||
| 95df7d3103 | |||
| 0d4673e910 | |||
| 03a818ae7f | |||
| c9bf3cde27 | |||
| d408523350 | |||
| e452d4425f | |||
| 003cad650a | |||
| 8da51f4251 | |||
| c4575a8e86 | |||
| 2fff59aa74 | |||
| 8523d6e882 | |||
| 3b4664de76 | |||
| 19f93fcd48 | |||
| c60aba26c4 | |||
| 747886efd6 | |||
| 4593c0a8df | |||
| 8e1a28d01e | |||
| 400daf51c3 | |||
| 6b37349ec6 | |||
| 72a7df8dd8 | |||
| fee8572163 | |||
| 7435ac6647 | |||
| 2237d71fa0 | |||
| ee3790cf1e | |||
| f382b3e921 | |||
| 4986175159 | |||
| da5d7bf9a9 | |||
| 634a74c442 | |||
| 247d03b23c | |||
| abb0ccdc52 | |||
| 52f5dfc51c | |||
| d5a819edbe | |||
| dae4b64c65 | |||
| aa3dd47b83 | |||
| 545e2a7027 | |||
| 2b67720d60 | |||
| e69b8dbb05 | |||
| 5079ca478c | |||
| ca74a45451 | |||
| 674962d1d6 | |||
| bd202704c9 | |||
| c7bf87a3ad | |||
| 2c36b78396 | |||
| 5eecb16a7e | |||
| 0409c0dd37 | |||
| 3de09f5840 | |||
| e40ce5a39b | |||
| cfd40e74d8 | |||
| 826ee659f0 | |||
| fa86ccd5da | |||
| 7f2be84a1f | |||
| a98388a768 | |||
| 283eafd2b0 | |||
| 22e7c74d19 | |||
| d23f05ec87 | |||
| 8ba6e3d90f | |||
| e5fc383b79 | |||
| 1b6955d3e9 | |||
| 791ca3ee09 | |||
| 02150cd3cf | |||
| 7e928a02bf | |||
| 361c8cde10 | |||
| 4382818579 | |||
| 3d59c588d5 | |||
| 2ed782963b | |||
| 02bfa2567f | |||
| e632c8241e | |||
| ce32db44e3 | |||
| e44770842a | |||
| 20bb3b2dd3 | |||
| e92bc480d5 | |||
| 48926e4c8f | |||
| ceb354b8a8 | |||
| 0335d4c1b9 | |||
| be99135caf | |||
| b25affb19a | |||
| 43041ff595 | |||
| 6869623d56 | |||
| 68be17dbcb | |||
| 05b812fef6 | |||
| ccbc33b32d | |||
| fc9b6b2421 | |||
| 1dd5f53628 | |||
| 857a172d13 | |||
| 232d452404 | |||
| bae4b62513 | |||
| 18e6d12420 | |||
| 520e7c8bd5 | |||
| a9a98232cf | |||
| 0308968e05 | |||
| 3016ad3045 | |||
| efb2969ec7 | |||
| aa5f0bec44 | |||
| fe78e8ecf0 | |||
| 0422ee9882 | |||
| f49e42632f | |||
| 77671c947e | |||
| df9a2634af | |||
| ab1c967cb5 | |||
| 170361fe97 | |||
| bf290e81a6 | |||
| e39942d8bc | |||
| e9f7bd20e8 | |||
| 705b144e6f | |||
| 3de069d235 | |||
| 341afc973a | |||
| 51233cd040 | |||
| 8cdbb973c1 | |||
| 4f98820f0f | |||
| 026b19fc79 | |||
| 3947dfc683 | |||
| 750f39bf14 | |||
| b1d6799d11 | |||
| 211827d36e | |||
| 2f508c9cd7 | |||
| d3810bf06d | |||
| d4d4334fe7 | |||
| db19c7bd76 | |||
| 6419ba7873 | |||
| 9897febe3e | |||
| 5a0f66135e | |||
| a957657e83 | |||
| 741dffbad8 | |||
| b36ae8d58f | |||
| d4cb86d7cc | |||
| 480e1c9e6c | |||
| 0442331b9b | |||
| 6bcc9982c6 | |||
| f40fa8ef47 | |||
| f2e0ccc27b | |||
| 426d0d35b2 | |||
| 418bd38e8c | |||
| 24ec55fef9 | |||
| 25f0cd2560 | |||
| c44c881932 | |||
| 63f3285606 | |||
| 701f64aea2 | |||
| 1a3e69ea25 | |||
| 09ebdb9867 | |||
| e515369be4 | |||
| e65503bfe9 | |||
| 40ed1032e2 | |||
| e26d547f57 | |||
| c62a48ed70 | |||
| ea14b85f8d | |||
| cecb665737 | |||
| 27cf8788d2 | |||
| 54d78f4b43 | |||
| 799c366915 | |||
| 4919faf809 | |||
| 1bcb416028 | |||
| c461eee56a | |||
| afa670dc85 | |||
| 1194f2f126 | |||
| 4bdab3560e | |||
| f029f67d73 | |||
| 91ec7ef871 | |||
| 1a97b127ec | |||
| e3507d0f34 | |||
| 9e9e44ae36 | |||
| 839b321a96 | |||
| f08c88a04c | |||
| fc1ae560da | |||
| ead1233f36 | |||
| d4e7761023 | |||
| e793b30940 | |||
| c5b1d85de0 | |||
| 4152e4e4a4 | |||
| e0e612664a | |||
| feb5e5e671 | |||
| d0ed84b269 | |||
| 99acbf88fb | |||
| a3e3bb15d3 | |||
| f62ef14617 | |||
| 25061f776a | |||
| eda4d23bb0 | |||
| cc58fb8190 | |||
| 0ebf06b334 | |||
| b64fc2582e | |||
| c744a155fd | |||
| 50a8c4fd73 | |||
| bc3a7cb01d | |||
| c2ba19c7c3 | |||
| 78ec0f3322 | |||
| 5cc20e638a | |||
| 81baa46f62 | |||
| 21f19a2a81 | |||
| 419cc318e7 | |||
| abf67d098e | |||
| 56a2ae4b20 | |||
| a45dbd049c | |||
| 58cd57f0a8 | |||
| 4865f4beb0 | |||
| 57bb9a43ac | |||
| c5e577c9fc | |||
| da1cf77ea7 | |||
| b4ab6405b4 | |||
| 18aa22256f | |||
| 34e0c249ed | |||
| 3ea210559a | |||
| 2d763eccd5 | |||
| e214baa8c2 | |||
| 9e9f5cc999 | |||
| 4828d5b2ef | |||
| 895136a19a | |||
| 90ee56839e | |||
| f68788d478 | |||
| 1bd0adfbea | |||
| b7c6e1080a | |||
| 157ae72a9c | |||
| e8707e1b9b | |||
| 5ea6378ec9 | |||
| 4d5c889efe | |||
| 0fe70c851b | |||
| e09f4284aa | |||
| 48924655f7 | |||
| 4c088fb4e3 | |||
| 0b672f9976 | |||
| f05168c76a | |||
| af13470d42 | |||
| e12f621011 | |||
| 85adad243b | |||
| 064f9317eb | |||
| 60806ad1f0 | |||
| e75368feed | |||
| 43d5fcd3fa | |||
| cb619b47ff | |||
| 028495841d | |||
| 10a9d8fd59 | |||
| 9aa1bda949 | |||
| 078098b48d | |||
| ef06c0d68a | |||
| 2c6851357e | |||
| 1a36738e99 | |||
| ec619c1e19 | |||
| bebeda0425 | |||
| 49c3dc45ed | |||
| ced76bd39a | |||
| f6c52fa870 | |||
| d322c502eb | |||
| 99bcf91ac1 | |||
| 8cb77ec6bf | |||
| e9a37778a1 | |||
| 3b6e465843 | |||
| e9c8f00ce8 | |||
| 96a58bd8c0 | |||
| 15ac5e9002 | |||
| b9b581a701 | |||
| 74c03395ba | |||
| 5a2068f6f9 | |||
| 7b06cdf0b6 | |||
| b7311d6449 | |||
| d0f2cbd9c5 | |||
| a4b4609844 | |||
| 027cf69b1d | |||
| c5be07a727 | |||
| dfe7e3d754 | |||
| 83d2ab521c | |||
| cbd105936a | |||
| bd8fb562f1 | |||
| f6d557ae29 | |||
| 0da0ba1de2 | |||
| 9f18b66134 | |||
| 18ada91129 | |||
| 8bb50c7414 | |||
| 01ebe441cc | |||
| 0c280b6c0f | |||
| bec4488439 | |||
| a2b72176e9 | |||
| cfbad0d0fa | |||
| 990a92e1de | |||
| 6ff6cfe686 | |||
| 0feab28222 | |||
| c6d56c0c14 | |||
| b417b22a6b | |||
| af72b4df5f | |||
| 9316b1bf3b | |||
| 7e058ec1b6 | |||
| 588bc0a2eb | |||
| 1e97e49910 | |||
| 708e73f0d0 | |||
| 7e5684450e | |||
| c9a065cbe4 | |||
| a41b67ba47 | |||
| d4def193d2 | |||
| c969d20f1e | |||
| 4df14a9911 | |||
| 087f434b9b | |||
| 097df9251d | |||
| 3cd3e4f1ac | |||
| 8da526055b | |||
| 4d7b25a689 | |||
| 04429c453f | |||
| d89f2a239d | |||
| 7ca63134c9 | |||
| c31e80d211 | |||
| 0b0d658e9c | |||
| 63615ecaac | |||
| 342c111ea2 | |||
| a181c2797a | |||
| 2eeb1c3218 | |||
| cdda7ab6c8 | |||
| 638ffe370e | |||
| dd8b9587c5 | |||
| b472bc6963 | |||
| 7a5abe1843 | |||
| f472e404ba | |||
| 50ea35fa48 | |||
| 966bc685ae | |||
| 55f1308300 | |||
| 3761e51b2c | |||
| 7957ec37c9 | |||
| 01e91d27aa | |||
| 2bf570daca | |||
| 7fd304ade6 | |||
| f11d123db6 | |||
| 05d29afc6a | |||
| cad9d801a2 | |||
| 40f595180d | |||
| 29533b085b | |||
| 2a4a90ad0e | |||
| d7141987af | |||
| 1203ffb4c5 | |||
| 465082e857 | |||
| 6ccc5ff748 | |||
| d605d913a3 | |||
| 5f55eb3745 | |||
| bc1bbe8925 | |||
| d6cf01e7ba | |||
| bca967c57c | |||
| 55c8c5124b | |||
| 1639dc7839 | |||
| c518c0af63 | |||
| af6a9f78b5 | |||
| cb80d66d8e | |||
| 834d695290 | |||
| b586d6036f | |||
| 3ee4c94896 | |||
| 972be2f290 | |||
| 6d4149d26a | |||
| 4ba5b1294d | |||
| 9b03f41936 | |||
| 8dc68faff7 | |||
| ed027d49e6 | |||
| f6536bcb63 | |||
| 1443257f4e | |||
| a10dc8fe3a | |||
| bdf765b201 | |||
| 89a20a5754 | |||
| df0c9c9384 | |||
| 9a2110db6f | |||
| 63a592c75a | |||
| d60b8b3fcb | |||
| bffcde9181 | |||
| 3190fe6714 | |||
| 12a9b5c017 | |||
| f0efe94ce4 | |||
| 49242c3eb5 | |||
| c4691c00b3 | |||
| c724f7fc6a | |||
| 2fd6a53612 | |||
| adaa4907fa | |||
| b5fa9d0915 | |||
| 8cc140aceb | |||
| 32d4105fdb | |||
| 5d0cd20370 | |||
| 682745fb2d | |||
| 4f0b65dcad | |||
| 71ec557996 | |||
| 5d0e22e883 | |||
| a17252c1d6 | |||
| 06db32735f | |||
| fbb04afa4f | |||
| 5774fd4d00 | |||
| 247642a68e | |||
| 7ebc2baec8 | |||
| 3ccfa48fa4 | |||
| a587f7e248 | |||
| ab2e3312f0 | |||
| b5311efb8f | |||
| d52225c83f | |||
| afdfe5a29d | |||
| e3d38d200c | |||
| 9835bd7b65 | |||
| 0cf30bba62 | |||
| 4236199533 | |||
| 13ded49cb8 | |||
| 83894d2124 | |||
| 15d2b141fe | |||
| 4bce6bda60 | |||
| 997cadae17 | |||
| 5943ad07a7 | |||
| 6ea1d9127b | |||
| 33de691447 | |||
| 2d6d957dd6 | |||
| 629e6d4113 | |||
| 6a040fed6c | |||
| 827fdc1e90 | |||
| 419443b508 | |||
| c611ff8eca | |||
| 8af5c0cb54 | |||
| 5c9927beb0 | |||
| 16c21a07f3 | |||
| bf98b65e69 | |||
| 9345568bb1 | |||
| c33adb645b | |||
| 5b6047fb5f | |||
| 6e5b36b1b9 | |||
| f7cbcddeb8 | |||
| a40116f80c | |||
| dc17a2c138 | |||
| 1e94faa9cb | |||
| bf196a939b | |||
| 1888564c23 | |||
| 60a6c08ded | |||
| d9ad883bdc | |||
| 84b80c578a | |||
| ccef53d6e3 | |||
| 4178a11d7c | |||
| 43c77cc198 | |||
| 93478b96f2 | |||
| a29e99a1b0 | |||
| 80d854f155 | |||
| 07c76f2501 | |||
| e611066329 | |||
| 47ba1b631a | |||
| efcb4a6c37 | |||
| 3462aa3948 | |||
| 655a452e30 | |||
| f9f34067a0 | |||
| 68d2494f17 | |||
| ac172d7d96 | |||
| 7865be81a9 | |||
| 3e158a03bc | |||
| 50daa293f0 | |||
| e30b965d35 | |||
| 0493e96590 | |||
| 33769eaa9f | |||
| 8e6d960262 | |||
| 7b7a7c0539 | |||
| 81d5445b8e | |||
| e1ff6bb141 | |||
| 3070f9f056 | |||
| 29733f0400 | |||
| 73b3b98e73 | |||
| e818b65b13 | |||
| 366e9ce065 | |||
| b4e687d48b | |||
| c13b45574d | |||
| 1f62d565cb | |||
| 308a7141bc | |||
| b2541eab1c | |||
| 6de4d2ad9d | |||
| 14ccf5792c | |||
| 9b169a971c | |||
| f9ab927d38 | |||
| e5d610c2fe | |||
| f90b3c0dde | |||
| f2d583bece | |||
| c303fcbd32 | |||
| 27f517ba9c | |||
| efdf82ebb0 | |||
| a95d402387 | |||
| 48cbef867d | |||
| 4372731867 | |||
| 3442e3211f | |||
| 8827920c4b | |||
| ce821c6c95 | |||
| b2062b3e93 | |||
| 8ef0b5b132 | |||
| e6df20c3b1 | |||
| 7abedbcbe9 | |||
| 2cbb34e3f6 | |||
| fd5db45e38 | |||
| 7f53ef65f8 | |||
| cf070c135a | |||
| 6c9c073e81 | |||
| 341e019f42 | |||
| eada806e0e | |||
| cc843ba43f | |||
| 1b93b85408 | |||
| f270f5cafd | |||
| 5c2008f080 | |||
| a307a26d45 | |||
| 8b5cfa19a2 | |||
| 63726ba60b | |||
| 25cf713c3a | |||
| 9be9c01d3d | |||
| 29e90b16ba | |||
| 8869e54f40 | |||
| 5a6da41712 | |||
| c65595f126 | |||
| 175453cff4 | |||
| 75e8618955 | |||
| 6103cca876 | |||
| 7f99c4c206 | |||
| 2f332328ac | |||
| 8cd6b1ac62 | |||
| 1d2f3497b1 | |||
| 8e26909e81 | |||
| ee9850f8bb | |||
| 92f07fe061 | |||
| b820fa01a3 | |||
| b87a4b8fa3 | |||
| 9c75c78a72 | |||
| 5d7eb69850 | |||
| 472e7197c7 | |||
| 130cff4c2a | |||
| 0656df7707 | |||
| b82f6d0b49 | |||
| dfd60fb10c | |||
| c47d2f0167 | |||
| bc1c2572ad | |||
| a39b90b074 | |||
| 31a77c0461 | |||
| ce64a73b4d | |||
| 273f488048 | |||
| 4466a1723e | |||
| a7f0e1f3ce | |||
| 26da4e5ead | |||
| 1813eb616c | |||
| 1ff6dfacd8 | |||
| 2d4ae3362c | |||
| 8e536c5d41 | |||
| f0d0b956cc | |||
| 516e9ca959 | |||
| 6845e55fac | |||
| 8f6b409ad0 | |||
| e7bab519c4 | |||
| 60a1514cb2 | |||
| d541cf9d17 | |||
| 0a25a690e2 | |||
| c2a0d0fb0b | |||
| 763a69b31b | |||
| 7c86a1add7 | |||
| d2cb893ef7 | |||
| 1e1f831f89 | |||
| ee396d9540 | |||
| 76681cfc7d | |||
| 5df90202a9 | |||
| aecd1def8e | |||
| bc911129b1 | |||
| 61fa2a7218 | |||
| 64a6224be3 | |||
| de2641b907 | |||
| 2d2af37fa2 | |||
| c099e6160a | |||
| 67e3fd54f9 | |||
| a93a46c4a1 | |||
| 0a28682107 | |||
| 3b1b20ca3d | |||
| 1f639fb7ea | |||
| c605d50b30 | |||
| 8a95c9290b | |||
| fc162a041a | |||
| aa8000dd1e | |||
| dc49ffa41d | |||
| 0f49236a6c | |||
| 7cb6aa92e2 | |||
| de22eb3c9c | |||
| 3167ed4ec1 | |||
| 427e173472 | |||
| 9c5f80ec23 | |||
| 564d3231d6 | |||
| 240df89774 | |||
| 959e0b8c07 | |||
| 21a1d22a83 | |||
| fc4b9a0017 | |||
| aedd07f814 | |||
| edba660840 | |||
| 391706bc0e | |||
| 591de02478 | |||
| e1ce676d46 | |||
| 440a0babb4 | |||
| 726afd8d20 | |||
| e33c5e90f8 | |||
| ad292c48e9 | |||
| 57568935f3 | |||
| 1a70959249 | |||
| 5bddc91935 | |||
| ece15889fa | |||
| 6359bd4a05 | |||
| 65196c44fd | |||
| 8b93cfdaa4 | |||
| c2a5c092a8 | |||
| b6254b71da | |||
| 180649670f | |||
| d6edb65a7a | |||
| a78d6bd111 | |||
| ad0e91ece7 | |||
| 66405520e2 |
+328
-35
@@ -2,41 +2,334 @@ kind: pipeline # 定义对象类型,还有secret和signature两种类型
|
||||
type: docker # 定义流水线类型,还有kubernetes、exec、ssh等类型
|
||||
name: cicd # 定义流水线名称
|
||||
|
||||
clone:
|
||||
disable: true
|
||||
|
||||
steps: # 定义流水线执行步骤,这些步骤将顺序执行
|
||||
- name: package # 流水线名称
|
||||
image: python:3.10-slim # 定义创建容器的Docker镜像
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: maven-cache
|
||||
path: /root/.m2 # 将maven下载依赖的目录挂载出来,防止重复下载
|
||||
- name: maven-build
|
||||
path: /app/build # 将应用打包好的Jar和执行脚本挂载出来
|
||||
commands: # 定义在Docker容器中执行的shell命令
|
||||
- pip install Cython
|
||||
- pip install wheel
|
||||
- pip install twine
|
||||
- cd ./src/bisheng-langchain
|
||||
- python setup.py bdist_wheel
|
||||
- cp dist.* /app/build/
|
||||
|
||||
- name: build_backend
|
||||
image: python:3.10-slim # 定义创建容器的Docker镜像
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: maven-cache
|
||||
path: /root/.m2 # 将maven下载依赖的目录挂载出来,防止重复下载
|
||||
- name: maven-build
|
||||
path: /app/build # 将应用打包好的Jar和执行脚本挂载出来
|
||||
- name: clone
|
||||
image: alpine/git
|
||||
pull: if-not-exists
|
||||
environment:
|
||||
http_proxy:
|
||||
from_secret: PROXY
|
||||
https_proxy:
|
||||
from_secret: PROXY
|
||||
commands:
|
||||
- cd ./src/backend
|
||||
- pip install bisheng_langchain==$RELEASE_VERSION
|
||||
- sed -i 's/^bisheng_langchain.*/bisheng_langchain = "'$RELEASE_VERSION'"/g' pyproject.toml
|
||||
- poetry lock
|
||||
- git config --global core.compression 0
|
||||
- git clone https://github.com/dataelement/bisheng.git .
|
||||
- git checkout $DRONE_COMMIT
|
||||
|
||||
- name: build-image # 步骤名称
|
||||
image: plugins/docker # 使用镜像
|
||||
settings: # 当前设置
|
||||
username: # 账号名称
|
||||
from_secret: docker_username
|
||||
password: # 账号密码
|
||||
from_secret: docker_password
|
||||
dockerfile: deploy/Dockerfile # Dockerfile地址, 注意是相对地址
|
||||
repo: yxs970707/deploy-web-demo # 镜像名称
|
||||
- name: set poetry
|
||||
pull: if-not-exists
|
||||
image: golang
|
||||
environment:
|
||||
RELEASE_VERSION: 99.99.99
|
||||
NEXUS_PUBLIC:
|
||||
from_secret: NEXUS_PUBLIC
|
||||
NEXUS_PUBLIC_PASSWORD:
|
||||
from_secret: NEXUS_PUBLIC_PASSWORD
|
||||
REPO:
|
||||
from_secret: PY_NEXUS
|
||||
PROXY:
|
||||
from_secret: APT-GET
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: bisheng-cache
|
||||
path: /app/build/
|
||||
commands:
|
||||
- cd ./src/backend
|
||||
- echo $REPO
|
||||
- REPO2=$(echo $REPO | sed 's/http:\\/\\///g')
|
||||
- sed '/apt-get/ s|$| '"$PROXY"'|' Dockerfile
|
||||
- sed -i '6i\RUN pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple' Dockerfile
|
||||
- sed -i '7i\RUN poetry source add --priority=supplemental foo http://'$NEXUS_PUBLIC':'$NEXUS_PUBLIC_PASSWORD'@'$REPO2'simple' Dockerfile
|
||||
- sed -i '8i\RUN poetry source add --priority=primary qh https://pypi.tuna.tsinghua.edu.cn/simple' Dockerfile
|
||||
- cat Dockerfile
|
||||
|
||||
- name: build_docker
|
||||
pull: if-not-exists
|
||||
image: docker:24.0.6
|
||||
privileged: true
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: apt-cache
|
||||
path: /var/cache/apt/archives # 将应用打包好的Jar和执行脚本挂载出来
|
||||
- name: socket
|
||||
path: /var/run/docker.sock
|
||||
- name: pro-cache
|
||||
path: /root/.local/share/pypoetry
|
||||
environment:
|
||||
http_proxy:
|
||||
from_secret: PROXY
|
||||
https_proxy:
|
||||
from_secret: PROXY
|
||||
no_proxy: 192.168.106.8
|
||||
version: release
|
||||
docker_registry: http://192.168.106.8:6082
|
||||
docker_repo: 192.168.106.8:6082/dataelement/bisheng-backend
|
||||
docker_user:
|
||||
from_secret: NEXUS_USER
|
||||
docker_password:
|
||||
from_secret: NEXUS_PASSWORD
|
||||
commands:
|
||||
- cd ./src/backend/
|
||||
- docker login -u $docker_user -p $docker_password $docker_registry
|
||||
- docker build -t $docker_repo:$version .
|
||||
- docker push $docker_repo:$version
|
||||
|
||||
- name: build_docker_frontend
|
||||
pull: if-not-exists
|
||||
image: docker:24.0.6
|
||||
privileged: true
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: apt-cache
|
||||
path: /var/cache/apt/archives # 将应用打包好的Jar和执行脚本挂载出来
|
||||
- name: socket
|
||||
path: /var/run/docker.sock
|
||||
environment:
|
||||
http_proxy:
|
||||
from_secret: PROXY
|
||||
https_proxy:
|
||||
from_secret: PROXY
|
||||
no_proxy: 192.168.106.8
|
||||
version: release
|
||||
docker_registry: http://192.168.106.8:6082
|
||||
docker_repo: 192.168.106.8:6082/dataelement/bisheng-frontend
|
||||
docker_user:
|
||||
from_secret: NEXUS_USER
|
||||
docker_password:
|
||||
from_secret: NEXUS_PASSWORD
|
||||
commands:
|
||||
- cd ./src/frontend/
|
||||
- docker login -u $docker_user -p $docker_password $docker_registry
|
||||
- docker build -t $docker_repo:$version .
|
||||
- docker push $docker_repo:$version
|
||||
|
||||
- name: ssh deploy
|
||||
image: appleboy/drone-ssh
|
||||
pull: if-not-exists
|
||||
settings:
|
||||
host: 192.168.106.116
|
||||
username: root
|
||||
password:
|
||||
from_secret: sshpwd
|
||||
script:
|
||||
- echo =======找到目录=======
|
||||
- cd /opt/server/bisheng-test
|
||||
- echo =======直接启动=======
|
||||
- docker compose pull
|
||||
- docker compose up -d
|
||||
|
||||
- name: notify-start # notify
|
||||
pull: if-not-exists
|
||||
image: plugins/webhook
|
||||
settings:
|
||||
debug: true
|
||||
urls:
|
||||
from_secret: FEISHU_URL
|
||||
content_type: application/json
|
||||
template: |
|
||||
{
|
||||
"msg_type": "interactive",
|
||||
"card": {
|
||||
"type": "template",
|
||||
"data": {
|
||||
"template_id": "AAqkI9bnY5FUs",
|
||||
"template_variable": {
|
||||
"repo_name": "{{ repo.name }}",
|
||||
"build_branch": "{{build.branch}}",
|
||||
"build_author": "{{ DRONE_COMMIT_AUTHOR }}",
|
||||
"link": "{{build.link}}",
|
||||
"commit_msg": "{{ trim build.message }}",
|
||||
"build_tag":"{{build.tag}}",
|
||||
"build_start":"{{build.started}}",
|
||||
"status": "{{ build.status }}"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
when: # 成功
|
||||
status:
|
||||
- success
|
||||
trigger:
|
||||
branch:
|
||||
- release
|
||||
event:
|
||||
- push
|
||||
|
||||
volumes:
|
||||
- name: bisheng-cache
|
||||
host:
|
||||
path: /opt/drone/data/bisheng/
|
||||
- name: pro-cache
|
||||
host:
|
||||
path: /opt/drone/data/pro/
|
||||
- name: apt-cache
|
||||
host:
|
||||
path: /opt/drone/data/bisheng/apt/
|
||||
- name: socket
|
||||
host:
|
||||
path: /var/run/docker.sock
|
||||
|
||||
|
||||
---
|
||||
|
||||
kind: pipeline # 定义对象类型,还有secret和signature两种类型
|
||||
type: docker # 定义流水线类型,还有kubernetes、exec、ssh等类型
|
||||
name: feat_cicd # 定义流水线名称
|
||||
|
||||
clone:
|
||||
disable: true
|
||||
|
||||
steps: # 定义流水线执行步骤,这些步骤将顺序执行
|
||||
- name: clone
|
||||
image: alpine/git
|
||||
pull: if-not-exists
|
||||
environment:
|
||||
http_proxy:
|
||||
from_secret: PROXY
|
||||
https_proxy:
|
||||
from_secret: PROXY
|
||||
commands:
|
||||
- git config --global core.compression 0
|
||||
- git clone https://github.com/dataelement/bisheng.git .
|
||||
- git checkout $DRONE_COMMIT
|
||||
|
||||
- name: set poetry
|
||||
pull: if-not-exists
|
||||
image: golang
|
||||
environment:
|
||||
NEXUS_PUBLIC:
|
||||
from_secret: NEXUS_PUBLIC
|
||||
NEXUS_PUBLIC_PASSWORD:
|
||||
from_secret: NEXUS_PUBLIC_PASSWORD
|
||||
REPO:
|
||||
from_secret: PY_NEXUS
|
||||
PROXY:
|
||||
from_secret: APT-GET
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: bisheng-cache
|
||||
path: /app/build/
|
||||
commands:
|
||||
- cd ./src/backend
|
||||
- echo $REPO
|
||||
- REPO2=$(echo $REPO | sed 's/http:\\/\\///g')
|
||||
- sed '/apt-get/ s|$| '"$PROXY"'|' Dockerfile
|
||||
- sed -i '6i\RUN pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple' Dockerfile
|
||||
- sed -i '7i\RUN poetry source add --priority=supplemental foo http://'$NEXUS_PUBLIC':'$NEXUS_PUBLIC_PASSWORD'@'$REPO2'simple' Dockerfile
|
||||
- sed -i '8i\RUN poetry source add --priority=primary qh https://pypi.tuna.tsinghua.edu.cn/simple' Dockerfile
|
||||
- cat Dockerfile
|
||||
|
||||
- name: build_docker
|
||||
pull: if-not-exists
|
||||
image: docker:24.0.6
|
||||
privileged: true
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: apt-cache
|
||||
path: /var/cache/apt/archives # 将应用打包好的Jar和执行脚本挂载出来
|
||||
- name: socket
|
||||
path: /var/run/docker.sock
|
||||
- name: pro-cache
|
||||
path: /root/.local/share/pypoetry
|
||||
environment:
|
||||
http_proxy:
|
||||
from_secret: PROXY
|
||||
https_proxy:
|
||||
from_secret: PROXY
|
||||
no_proxy: 192.168.106.8
|
||||
version: ${DRONE_BRANCH}
|
||||
docker_registry: http://192.168.106.8:6082
|
||||
docker_repo: 192.168.106.8:6082/dataelement/bisheng-backend
|
||||
docker_user:
|
||||
from_secret: NEXUS_USER
|
||||
docker_password:
|
||||
from_secret: NEXUS_PASSWORD
|
||||
commands:
|
||||
- echo "old tag is $version"
|
||||
- version=$(echo $version | sed 's/\\//_/g')
|
||||
- echo "build image tag is $version"
|
||||
- cd ./src/backend/
|
||||
- docker login -u $docker_user -p $docker_password $docker_registry
|
||||
- docker build -t $docker_repo:$version .
|
||||
- docker push $docker_repo:$version
|
||||
|
||||
- name: build_docker_frontend
|
||||
pull: if-not-exists
|
||||
image: docker:24.0.6
|
||||
privileged: true
|
||||
volumes: # 将容器内目录挂载到宿主机,仓库需要开启Trusted设置
|
||||
- name: apt-cache
|
||||
path: /var/cache/apt/archives # 将应用打包好的Jar和执行脚本挂载出来
|
||||
- name: socket
|
||||
path: /var/run/docker.sock
|
||||
environment:
|
||||
http_proxy:
|
||||
from_secret: PROXY
|
||||
https_proxy:
|
||||
from_secret: PROXY
|
||||
no_proxy: 192.168.106.8
|
||||
version: ${DRONE_BRANCH}
|
||||
docker_registry: http://192.168.106.8:6082
|
||||
docker_repo: 192.168.106.8:6082/dataelement/bisheng-frontend
|
||||
docker_user:
|
||||
from_secret: NEXUS_USER
|
||||
docker_password:
|
||||
from_secret: NEXUS_PASSWORD
|
||||
commands:
|
||||
- echo "old tag is $version"
|
||||
- version=$(echo $version | sed 's/\\//_/g')
|
||||
- echo "build image tag is $version"
|
||||
- cd ./src/frontend/
|
||||
- docker login -u $docker_user -p $docker_password $docker_registry
|
||||
- docker build -t $docker_repo:$version .
|
||||
- docker push $docker_repo:$version
|
||||
|
||||
- name: notify-start # notify
|
||||
pull: if-not-exists
|
||||
image: plugins/webhook
|
||||
settings:
|
||||
debug: true
|
||||
urls:
|
||||
from_secret: FEISHU_URL
|
||||
content_type: application/json
|
||||
template: |
|
||||
{
|
||||
"msg_type": "interactive",
|
||||
"card": {
|
||||
"type": "template",
|
||||
"data": {
|
||||
"template_id": "AAqkI9bnY5FUs",
|
||||
"template_variable": {
|
||||
"repo_name": "{{ repo.name }}",
|
||||
"build_branch": "{{build.branch}}",
|
||||
"build_author": "{{ DRONE_COMMIT_AUTHOR }}",
|
||||
"link": "{{build.link}}",
|
||||
"commit_msg": "{{ trim build.message }}",
|
||||
"build_tag":"{{build.tag}}",
|
||||
"build_start":"{{build.started}}",
|
||||
"status": "{{ build.status }}"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
when: # 成功
|
||||
status:
|
||||
- success
|
||||
trigger:
|
||||
branch:
|
||||
- add_some_branch_you_need
|
||||
- chore/fix-vulnerability
|
||||
event:
|
||||
- push
|
||||
|
||||
volumes:
|
||||
- name: bisheng-cache
|
||||
host:
|
||||
path: /opt/drone/data/bisheng/
|
||||
- name: pro-cache
|
||||
host:
|
||||
path: /opt/drone/data/pro/
|
||||
- name: apt-cache
|
||||
host:
|
||||
path: /opt/drone/data/bisheng/apt/
|
||||
- name: socket
|
||||
host:
|
||||
path: /var/run/docker.sock
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
# 默认:自动识别文本,统一用 LF 存库
|
||||
* text=auto eol=lf
|
||||
|
||||
# 明确常见文本文件用 LF
|
||||
*.py text eol=lf
|
||||
*.sh text eol=lf
|
||||
*.yml text eol=lf
|
||||
*.yaml text eol=lf
|
||||
*.md text eol=lf
|
||||
*.txt text eol=lf
|
||||
*.json text eol=lf
|
||||
*.toml text eol=lf
|
||||
*.cfg text eol=lf
|
||||
*.ini text eol=lf
|
||||
|
||||
# Windows 脚本保留 CRLF
|
||||
*.bat text eol=crlf
|
||||
*.cmd text eol=crlf
|
||||
|
||||
# 二进制:禁止任何换行转换和 diff
|
||||
*.png binary
|
||||
*.jpg binary
|
||||
*.jpeg binary
|
||||
*.gif binary
|
||||
*.ico binary
|
||||
*.pdf binary
|
||||
*.zip binary
|
||||
*.tar binary
|
||||
*.gz binary
|
||||
*.7z binary
|
||||
*.mp4 binary
|
||||
*.docx binary
|
||||
*.xlsx binary
|
||||
*.pptx binary
|
||||
@@ -0,0 +1,127 @@
|
||||
name: BASE_CI
|
||||
|
||||
on:
|
||||
push:
|
||||
# Sequence of patterns matched against refs/tags
|
||||
tags:
|
||||
- "base.v*"
|
||||
|
||||
env:
|
||||
DOCKERHUB_REPO: dataelement/
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
build_bisheng_arm:
|
||||
runs-on: ubuntu-latest
|
||||
# if: startsWith(github.event.ref, 'refs/tags')
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=${{ steps.get_version.outputs.VERSION }}" >> $GITHUB_ENV
|
||||
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# 构建 backend 并推送到 Docker hub
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v1
|
||||
|
||||
- name: set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Build backend arm64 and push
|
||||
id: docker_build_backend
|
||||
run: |
|
||||
docker buildx build --build-arg PANDOC_ARCH=arm64 --file ./src/backend/base.Dockerfile --platform linux/arm64 --provenance false --tag ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-arm64 --push ./src/backend/
|
||||
|
||||
build_bisheng_amd:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=${{ steps.get_version.outputs.VERSION }}" >> $GITHUB_ENV
|
||||
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Build backend amd64 and push
|
||||
id: docker_build_backend
|
||||
run: |
|
||||
docker buildx build --build-arg PANDOC_ARCH=amd64 --file ./src/backend/base.Dockerfile --platform linux/amd64 --provenance false --tag ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64 --push ./src/backend/
|
||||
|
||||
|
||||
combine_two_images:
|
||||
runs-on: ubuntu-latest
|
||||
needs:
|
||||
- build_bisheng_amd
|
||||
- build_bisheng_arm
|
||||
steps:
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=${{ steps.get_version.outputs.VERSION }}" >> $GITHUB_ENV
|
||||
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
- name: Combine Two images
|
||||
run: |
|
||||
docker manifest create ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }} ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-arm64 ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
docker manifest push ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
# 获取提交信息
|
||||
- name: Process git message
|
||||
id: process_message
|
||||
run: |
|
||||
value=$(echo "${{ github.event.head_commit.message }}" | sed -e ':a' -e 'N' -e '$!ba' -e 's/\n/%0A/g')
|
||||
value=$(echo "${value}" | sed -e ':a' -e 'N' -e '$!ba' -e 's/\r/%0A/g')
|
||||
echo "message=${value}" >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
# 飞书通知
|
||||
- name: notify feishu
|
||||
uses: fjogeleit/http-request-action@v1
|
||||
with:
|
||||
url: ${{ secrets.FEISHU_WEBHOOK }}
|
||||
method: 'POST'
|
||||
data: '{"msg_type":"post","content":{"post":{"zh_cn":{"title": "${{ steps.get_version.outputs.VERSION }}发布成功", "content": [[{"tag":"text","text":"基础镜像"},{"tag":"text","text":"${{ env.message }}"}]]}}}}'
|
||||
+108
-77
@@ -14,43 +14,82 @@ concurrency:
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
build_bisheng_langchain:
|
||||
|
||||
build_bisheng_backend:
|
||||
runs-on: ubuntu-latest
|
||||
#if: startsWith(github.event.ref, 'refs/tags')
|
||||
# if: startsWith(github.event.ref, 'refs/tags')
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=${{ steps.get_version.outputs.VERSION }}" >> $GITHUB_ENV
|
||||
run: echo "RELEASE_VERSION=1.3.1" >> $GITHUB_ENV
|
||||
|
||||
# 构建 bisheng_langchain
|
||||
- name: Set python version 3.8
|
||||
uses: actions/setup-python@v1
|
||||
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
python-version: 3.8
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# 构建 backend 并推送到 Docker hub
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v1
|
||||
|
||||
- name: Build PyPi bisheng-langchain and push
|
||||
id: pypi_build_bisheng_langchain
|
||||
- name: set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Build backend and push
|
||||
id: docker_build_backend
|
||||
run: |
|
||||
pip install Cython
|
||||
pip install wheel
|
||||
pip install twine
|
||||
cd ./src/bisheng-langchain
|
||||
python setup.py bdist_wheel
|
||||
set +e
|
||||
twine upload dist/* -u ${{ secrets.PYPI_USER }} -p ${{ secrets.PYPI_PASSWORD }} --repository pypi
|
||||
set -e
|
||||
|
||||
build_bisheng:
|
||||
needs: build_bisheng_langchain
|
||||
docker buildx build --file ./src/backend/Dockerfile --platform linux/amd64 --provenance false --tag ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64 --push ./src/backend/
|
||||
|
||||
build_backend_arm:
|
||||
runs-on: ubuntu-22.04-arm
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=1.3.1" >> $GITHUB_ENV
|
||||
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# - name: Set up QEMU
|
||||
# uses: docker/setup-qemu-action@v1
|
||||
|
||||
- name: set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Build backend and push
|
||||
id: docker_build_backend
|
||||
run: |
|
||||
docker buildx build --file ./src/backend/Dockerfile --platform linux/arm64 --provenance false --tag ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-arm64 --push ./src/backend/
|
||||
|
||||
build_bisheng_frontend:
|
||||
runs-on: ubuntu-latest
|
||||
# if: startsWith(github.event.ref, 'refs/tags')
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
@@ -73,68 +112,59 @@ jobs:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# - name: Login to DockerHub
|
||||
# uses: docker/login-action@v1
|
||||
# with:
|
||||
# registry: https://cr.dataelem.com/
|
||||
# username: ${{ secrets.CR_DOCKERHUB_USERNAME }}
|
||||
# password: ${{ secrets.CR_DOCKERHUB_TOKEN }}
|
||||
|
||||
# 构建 backend 并推送到 Docker hub
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v1
|
||||
- name: Build frontend and push
|
||||
id: docker_build_frontend
|
||||
run: |
|
||||
docker buildx build --file ./src/frontend/Dockerfile --platform linux/amd64 --provenance false --tag ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-amd64 --push ./src/frontend/
|
||||
|
||||
build_frontend_arm:
|
||||
runs-on: ubuntu-22.04-arm
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=${{ steps.get_version.outputs.VERSION }}" >> $GITHUB_ENV
|
||||
|
||||
# - name: Set up QEMU
|
||||
# uses: docker/setup-qemu-action@v1
|
||||
|
||||
- name: set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: install poetry
|
||||
uses: snok/install-poetry@v1
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
installer-parallel: true
|
||||
- name: build lock
|
||||
run: |
|
||||
cd ./src/backend
|
||||
pip install bisheng_langchain==$RELEASE_VERSION
|
||||
sed -i 's/^bisheng_langchain.*/bisheng_langchain = "'$RELEASE_VERSION'"/g' pyproject.toml
|
||||
poetry lock
|
||||
cd ../../
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Build backend and push
|
||||
id: docker_build_backend
|
||||
uses: docker/build-push-action@v2
|
||||
with:
|
||||
# backend 的context目录
|
||||
context: "./src/backend/"
|
||||
# 是否 docker push
|
||||
push: true
|
||||
# docker build arg, 注入 APP_NAME/APP_VERSION
|
||||
platforms: linux/amd64,linux/arm64
|
||||
build-args: |
|
||||
APP_NAME="bisheng-backend"
|
||||
APP_VERSION=${{ steps.get_version.outputs.VERSION }}
|
||||
# 生成两个 docker tag: ${APP_VERSION} 和 latest
|
||||
tags: |
|
||||
${{ env.DOCKERHUB_REPO }}bisheng-backend:latest
|
||||
${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
# 构建 Docker frontend 并推送到 Docker hub
|
||||
- name: Build frontend and push
|
||||
id: docker_build_frontend
|
||||
uses: docker/build-push-action@v2
|
||||
with:
|
||||
# frontend 的context目录
|
||||
context: "./src/frontend/"
|
||||
# 是否 docker push
|
||||
push: true
|
||||
# docker build arg, 注入 APP_NAME/APP_VERSION
|
||||
platforms: linux/amd64,linux/arm64
|
||||
build-args: |
|
||||
APP_NAME="bisheng-frontend"
|
||||
APP_VERSION=${{ steps.get_version.outputs.VERSION }}
|
||||
# 生成两个 docker tag: ${APP_VERSION} 和 latest
|
||||
tags: |
|
||||
${{ env.DOCKERHUB_REPO }}bisheng-frontend:latest
|
||||
${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
run: |
|
||||
docker buildx build --file ./src/frontend/Dockerfile --platform linux/arm64 --provenance false --tag ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-arm64 --push ./src/frontend/
|
||||
|
||||
notify_feishu:
|
||||
needs:
|
||||
- build_bisheng_backend
|
||||
- build_backend_arm
|
||||
- build_bisheng_frontend
|
||||
- build_frontend_arm
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Process git message
|
||||
id: process_message
|
||||
run: |
|
||||
@@ -148,4 +178,5 @@ jobs:
|
||||
with:
|
||||
url: ${{ secrets.FEISHU_WEBHOOK }}
|
||||
method: 'POST'
|
||||
data: '{"msg_type":"post","content":{"post":{"zh_cn":{"title": "${{ steps.get_version.outputs.VERSION }}发布成功", "content": [[{"tag":"text","text":"发布功能:"},{"tag":"text","text":"${{ env.message }}"}]]}}}}'
|
||||
data: '{"msg_type":"post","content":{"post":{"zh_cn":{"title": "${{ steps.get_version.outputs.VERSION }}-amd64镜像预发布成功", "content": [[{"tag":"text","text":"发布功能:"},{"tag":"text","text":"${{ env.message }}"}]]}}}}'
|
||||
|
||||
+132
-119
@@ -1,143 +1,156 @@
|
||||
name: release
|
||||
name: PublishRelease
|
||||
|
||||
# 在github上新建release发行版时触发此CICD,主要是把预发布镜像的tag改为正式镜像的tag,并同步到私有镜像仓库
|
||||
on:
|
||||
push:
|
||||
# Sequence of patterns matched against refs/tags
|
||||
branches:
|
||||
- "release"
|
||||
release:
|
||||
types: [published]
|
||||
|
||||
env:
|
||||
DOCKERHUB_REPO: dataelement/
|
||||
PY_NEXUS: 110.16.193.170:50083
|
||||
DOCKER_NEXUS: 110.16.193.170:50080
|
||||
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
build:
|
||||
combine_publish_images:
|
||||
runs-on: ubuntu-latest
|
||||
#if: startsWith(github.event.ref, 'refs/tags')
|
||||
steps:
|
||||
# deploy
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Process git message
|
||||
id: process_message
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
value=$(echo "${{ github.event.head_commit.message }}" | sed -e ':a' -e 'N' -e '$!ba' -e 's/\n/%0A/g' )
|
||||
echo "message=${value}" >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
- name: notify feishu
|
||||
uses: fjogeleit/http-request-action@v1
|
||||
with:
|
||||
url: ' https://open.feishu.cn/open-apis/bot/v2/hook/2cfe0d8d-647c-4408-9f39-c59134035c4b'
|
||||
method: 'POST'
|
||||
data: '{"msg_type":"post","content":{"post":{"zh_cn":{"title": "${{github.event.pusher.name}}提交代码,开始编译", "content": [[{"tag":"text","text":"发布功能:"},{"tag":"text","text":"${{ env.message }}"}]]}}}}'
|
||||
|
||||
- name: Set Environment Variable
|
||||
run: echo "RELEASE_VERSION=99.99.99" >> $GITHUB_ENV
|
||||
|
||||
- name: Set python version 3.8
|
||||
uses: actions/setup-python@v1
|
||||
with:
|
||||
python-version: 3.8
|
||||
|
||||
- name: Build PyPi bisheng-langchain and push
|
||||
id: pypi_build_bisheng_langchain
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
- name: Echo version
|
||||
id: echo_version
|
||||
run: |
|
||||
pip install Cython
|
||||
pip install wheel
|
||||
pip install twine
|
||||
cd ./src/bisheng-langchain
|
||||
python setup.py bdist_wheel
|
||||
repo="http://${{ env.PY_NEXUS }}/repository/pypi-hosted/"
|
||||
twine upload --verbose -u ${{ secrets.NEXUS_USER }} -p ${{ secrets.NEXUS_PASSWORD }} --repository-url $repo dist/*.whl
|
||||
cd ../../
|
||||
echo "this release is link version: ${{ steps.get_version.outputs.VERSION }}"
|
||||
|
||||
# 发布到 私有仓库
|
||||
- name: set insecure registry
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Combine two images
|
||||
id: combine_two_images
|
||||
run: |
|
||||
echo "{ \"insecure-registries\": [\"http://${{ env.DOCKER_NEXUS }}\"] }" | sudo tee /etc/docker/daemon.json
|
||||
sudo service docker restart
|
||||
docker manifest create ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }} ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-arm64 ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
docker manifest push ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
# - name: Set up QEMU
|
||||
# uses: docker/setup-qemu-action@v1
|
||||
docker manifest create ${{ env.DOCKERHUB_REPO }}bisheng-backend:latest ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-arm64 ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
docker manifest push ${{ env.DOCKERHUB_REPO }}bisheng-backend:latest
|
||||
|
||||
- name: Login Nexus Container Registry
|
||||
uses: docker/login-action@v2
|
||||
with:
|
||||
registry: http://${{ env.DOCKER_NEXUS }}/
|
||||
username: ${{ secrets.NEXUS_USER }}
|
||||
password: ${{ secrets.NEXUS_PASSWORD }}
|
||||
docker manifest create ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }} ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-arm64 ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
docker manifest push ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
# 替换poetry编译为私有服务
|
||||
- name: replace self-host repo
|
||||
uses: snok/install-poetry@v1
|
||||
with:
|
||||
installer-parallel: true
|
||||
|
||||
- name: build lock
|
||||
docker manifest create ${{ env.DOCKERHUB_REPO }}bisheng-frontend:latest ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-arm64 ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
docker manifest push ${{ env.DOCKERHUB_REPO }}bisheng-frontend:latest
|
||||
|
||||
sync_dataelem_repos:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
cd ./src/backend
|
||||
sed -i 's/^bisheng_langchain.*/bisheng_langchain = "'$RELEASE_VERSION'"/g' pyproject.toml
|
||||
poetry source add --priority=supplemental foo http://${{ secrets.NEXUS_PUBLIC }}:${{ secrets.NEXUS_PUBLIC_PASSWORD }}@${{ env.PY_NEXUS }}/repository/pypi-group/simple
|
||||
poetry lock
|
||||
cd ../../
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
- name: Echo version
|
||||
id: echo_version
|
||||
run: |
|
||||
echo "this release is link version: ${{ steps.get_version.outputs.VERSION }}"
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
registry: https://cr.dataelem.com/
|
||||
username: ${{ secrets.CR_DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.CR_DOCKERHUB_TOKEN }}
|
||||
|
||||
# 构建 backend 并推送到 Docker hub
|
||||
- name: Build backend and push
|
||||
id: docker_build_backend
|
||||
uses: docker/build-push-action@v2
|
||||
with:
|
||||
# backend 的context目录
|
||||
context: "./src/backend/"
|
||||
# 是否 docker push
|
||||
push: true
|
||||
# docker build arg, 注入 APP_NAME/APP_VERSION
|
||||
build-args: |
|
||||
APP_NAME="bisheng-backend"
|
||||
APP_VERSION="release"
|
||||
# 生成两个 docker tag: ${APP_VERSION} 和 latest
|
||||
tags: |
|
||||
${{ env.DOCKER_NEXUS }}/${{ env.DOCKERHUB_REPO }}bisheng-backend:release
|
||||
# 构建 Docker frontend 并推送到 Docker hub
|
||||
- name: Build frontend and push
|
||||
id: docker_build_frontend
|
||||
uses: docker/build-push-action@v2
|
||||
with:
|
||||
# frontend 的context目录
|
||||
context: "./src/frontend/"
|
||||
# 是否 docker push
|
||||
push: true
|
||||
# docker build arg, 注入 APP_NAME/APP_VERSION
|
||||
build-args: |
|
||||
APP_NAME="bisheng-frontend"
|
||||
APP_VERSION="release"
|
||||
# 生成两个 docker tag: ${APP_VERSION} 和 latest
|
||||
tags: |
|
||||
${{ env.DOCKER_NEXUS }}/${{ env.DOCKERHUB_REPO }}bisheng-frontend:release
|
||||
# deploy
|
||||
- name: notify feishu
|
||||
uses: fjogeleit/http-request-action@v1
|
||||
with:
|
||||
url: ' https://open.feishu.cn/open-apis/bot/v2/hook/2cfe0d8d-647c-4408-9f39-c59134035c4b'
|
||||
method: 'POST'
|
||||
data: '{"msg_type":"text","content":{"text":"release 编译成功, 准备部署"}}'
|
||||
- name: Sync images
|
||||
id: sync_images
|
||||
run: |
|
||||
echo "sync backend images"
|
||||
docker pull ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
|
||||
docker tag ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64 cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
docker tag ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}-amd64 cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-backend:latest
|
||||
|
||||
docker push cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
docker push cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-backend:latest
|
||||
|
||||
- name: Deploy Stage
|
||||
uses: fjogeleit/http-request-action@v1
|
||||
echo "sync frontend images"
|
||||
docker pull ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-amd64
|
||||
|
||||
docker tag ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-amd64 cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
docker tag ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}-amd64 cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-frontend:latest
|
||||
|
||||
docker push cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
docker push cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-frontend:latest
|
||||
echo "--- sync over ---"
|
||||
|
||||
test_pull_images:
|
||||
needs:
|
||||
- combine_publish_images
|
||||
- sync_dataelem_repos
|
||||
runs-on: ubuntu-22.04
|
||||
steps:
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
- name: Echo version
|
||||
id: echo_version
|
||||
run: |
|
||||
echo "this release is link version: ${{ steps.get_version.outputs.VERSION }}"
|
||||
# 登录 cr docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
timeout: 200000
|
||||
url: 'https://bisheng.dataelem.com/deploy/cgi-bin/deploy_script.py'
|
||||
method: 'GET'
|
||||
|
||||
- name: notify feishu
|
||||
uses: fjogeleit/http-request-action@v1
|
||||
registry: https://cr.dataelem.com/
|
||||
username: ${{ secrets.CR_DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.CR_DOCKERHUB_TOKEN }}
|
||||
# 登录 docker hub
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v1
|
||||
with:
|
||||
url: ' https://open.feishu.cn/open-apis/bot/v2/hook/2cfe0d8d-647c-4408-9f39-c59134035c4b'
|
||||
method: 'POST'
|
||||
data: '{"msg_type":"text","content":{"text":"release 部署成功"}}'
|
||||
# GitHub Repo => Settings => Secrets 增加 docker hub 登录密钥信息
|
||||
# DOCKERHUB_USERNAME 是 docker hub 账号名.
|
||||
# DOCKERHUB_TOKEN: docker hub => Account Setting => Security 创建.
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
- name: Test pull images
|
||||
run: |
|
||||
docker pull ${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
docker pull cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
docker pull ${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
docker pull cr.dataelem.com/${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
|
||||
notify_feishu:
|
||||
needs:
|
||||
- test_pull_images
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/tags\//}
|
||||
|
||||
- name: Process git message
|
||||
id: process_message
|
||||
run: |
|
||||
value=$(echo "${{ github.event.head_commit.message }}" | sed -e ':a' -e 'N' -e '$!ba' -e 's/\n/%0A/g')
|
||||
value=$(echo "${value}" | sed -e ':a' -e 'N' -e '$!ba' -e 's/\r/%0A/g')
|
||||
echo "message=${value}" >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
- name: notify feishu
|
||||
uses: fjogeleit/http-request-action@v1
|
||||
with:
|
||||
url: ${{ secrets.FEISHU_WEBHOOK }}
|
||||
method: 'POST'
|
||||
data: '{"msg_type":"post","content":{"post":{"zh_cn":{"title": "${{ steps.get_version.outputs.VERSION }}镜像发布成功", "content": [[{"tag":"text","text":"发布功能:"},{"tag":"text","text":"${{ env.message }}"}]]}}}}'
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
name: test_build
|
||||
|
||||
on:
|
||||
push:
|
||||
# Sequence of patterns matched against refs/tags
|
||||
branches:
|
||||
- "develop/*"
|
||||
|
||||
env:
|
||||
DOCKERHUB_REPO: project/
|
||||
PY_NEXUS: 110.16.193.170:50083
|
||||
DOCKER_NEXUS: 110.16.193.170:50080
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
#if: startsWith(github.event.ref, 'refs/tags')
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Get version
|
||||
id: get_version
|
||||
run: |
|
||||
echo ::set-output name=VERSION::${GITHUB_REF/refs\/heads\/develop\//}
|
||||
echo $GITHUB_REF
|
||||
echo $VERSION
|
||||
|
||||
# 构建 bisheng-langchain
|
||||
- name: Set python version 3.8
|
||||
uses: actions/setup-python@v1
|
||||
with:
|
||||
python-version: 3.8
|
||||
|
||||
# 发布到 私有仓库
|
||||
- name: set insecure registry
|
||||
run: |
|
||||
echo "{ \"insecure-registries\": [\"http://${{ env.DOCKER_NEXUS }}\"] }" | sudo tee /etc/docker/daemon.json
|
||||
sudo service docker restart
|
||||
|
||||
# - name: Set up QEMU
|
||||
# uses: docker/setup-qemu-action@v1
|
||||
|
||||
- name: Login Nexus Container Registry
|
||||
uses: docker/login-action@v2
|
||||
with:
|
||||
registry: http://${{ env.DOCKER_NEXUS }}/
|
||||
username: ${{ secrets.NEXUS_USER }}
|
||||
password: ${{ secrets.NEXUS_PASSWORD }}
|
||||
|
||||
# 替换poetry编译为私有服务
|
||||
- name: replace self-host repo
|
||||
uses: snok/install-poetry@v1
|
||||
with:
|
||||
installer-parallel: true
|
||||
|
||||
- name: build lock
|
||||
run: |
|
||||
cd ./src/backend
|
||||
poetry source add --priority=supplemental foo http://${{ secrets.NEXUS_PUBLIC }}:${{ secrets.NEXUS_PUBLIC_PASSWORD }}@${{ env.PY_NEXUS }}/repository/pypi-group/simple
|
||||
poetry lock
|
||||
cd ../../
|
||||
|
||||
# 构建 backend 并推送到 Docker hub
|
||||
- name: Build backend and push
|
||||
id: docker_build_backend
|
||||
uses: docker/build-push-action@v2
|
||||
with:
|
||||
# backend 的context目录
|
||||
context: "./src/backend/"
|
||||
# 是否 docker push
|
||||
push: true
|
||||
# docker build arg, 注入 APP_NAME/APP_VERSION
|
||||
build-args: |
|
||||
APP_NAME="bisheng-backend"
|
||||
APP_VERSION=${{ steps.get_version.outputs.VERSION }}
|
||||
# 生成两个 docker tag: ${APP_VERSION} 和 latest
|
||||
tags: |
|
||||
${{ env.DOCKER_NEXUS }}/${{ env.DOCKERHUB_REPO }}bisheng-backend:${{ steps.get_version.outputs.VERSION }}
|
||||
# 构建 Docker frontend 并推送到 Docker hub
|
||||
- name: Build frontend and push
|
||||
id: docker_build_frontend
|
||||
uses: docker/build-push-action@v2
|
||||
with:
|
||||
# frontend 的context目录
|
||||
context: "./src/frontend/"
|
||||
# 是否 docker push
|
||||
push: true
|
||||
# docker build arg, 注入 APP_NAME/APP_VERSION
|
||||
build-args: |
|
||||
APP_NAME="bisheng-frontend"
|
||||
APP_VERSION=${{ steps.get_version.outputs.VERSION }}
|
||||
# 生成两个 docker tag: ${APP_VERSION} 和 latest
|
||||
tags: |
|
||||
${{ env.DOCKER_NEXUS }}/${{ env.DOCKERHUB_REPO }}bisheng-frontend:${{ steps.get_version.outputs.VERSION }}
|
||||
|
||||
+7
-5
@@ -79,7 +79,7 @@ typings/
|
||||
.node_repl_history
|
||||
|
||||
# Output of 'npm pack'
|
||||
*.tgz
|
||||
# *.tgz
|
||||
|
||||
# Yarn Integrity file
|
||||
.yarn-integrity
|
||||
@@ -96,7 +96,6 @@ typings/
|
||||
|
||||
# Nuxt.js build / generate output
|
||||
.nuxt
|
||||
dist
|
||||
|
||||
# Gatsby files
|
||||
.cache/
|
||||
@@ -134,13 +133,11 @@ build/
|
||||
output/
|
||||
develop-eggs/
|
||||
config.dev.yaml
|
||||
dist/
|
||||
downloads/
|
||||
eggs/
|
||||
.eggs/
|
||||
lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
wheels/
|
||||
@@ -251,7 +248,7 @@ dmypy.json
|
||||
|
||||
# Poetry
|
||||
.testenv/*
|
||||
|
||||
poetry.lock
|
||||
|
||||
.githooks/prepare-commit-msg
|
||||
.langchain.db
|
||||
@@ -263,3 +260,8 @@ sftp-config.json
|
||||
|
||||
/tmp/*
|
||||
sftp-config.json
|
||||
|
||||
# Docker local files
|
||||
docker/data/*
|
||||
docker/mysql/data/*
|
||||
docker/office/bisheng/*.gz
|
||||
@@ -175,18 +175,7 @@
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright 2022 Dataelement Technologies, Inc
|
||||
Copyright © 2024 Dataelement Technologies, Inc
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
@@ -199,38 +188,3 @@
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
|
||||
The Bisheng is licensed under the Apache License 2.0, with the following additional conditions:
|
||||
|
||||
1. Bisheng is permitted to be used for commercialization. You can use Bisheng as a "backend-as-a-service" for your other applications, or deliver it to enterprises as an application development platform. However, when the following conditions are met, you must contact the producer to obtain a commercial license:
|
||||
|
||||
a. Multi-tenant SaaS service: Unless explicitly authorized by Bisheng in writing, you may not use the Bisheng source code to operate a multi-tenant SaaS service that is similar to the Bisheng.
|
||||
b. LOGO and copyright information: In the process of using Bisheng, you may not remove or replace the LOGO or copyright information in the Bisheng console.
|
||||
|
||||
Please contact hanfeng@dataelem.com by email to inquire about licensing matters.
|
||||
|
||||
2. As a contributor, you should agree that your contributed code:
|
||||
|
||||
a. The producer can adjust the open-source agreement to be more strict or relaxed.
|
||||
b. Can be used for commercial purposes, such as Bisheng's cloud business.
|
||||
|
||||
Apart from this, all other rights and restrictions follow the Apache License 2.0. If you need more detailed information, you can refer to the full version of Apache License 2.0.
|
||||
|
||||
The interactive design of this product is protected by an appearance patent.
|
||||
|
||||
© 2023 Bisheng.
|
||||
|
||||
---
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
@@ -1,19 +1,24 @@
|
||||
<img src="https://www.dataelem.com/nstatic/bisheng.png" alt="Bisheng banner">
|
||||
**Proudly made by Chinese,May we, like the creators of Deepseek and Black Myth: Wukong, bring more wonder and greatness to the world.**
|
||||
|
||||
> 源自中国匠心,希望我们能像 [Deepseek]、[黑神话:悟空] 团队一样,给世界带来更多美好。
|
||||
|
||||
<img src="https://dataelem.com/bs/face.png" alt="Bisheng banner">
|
||||
|
||||
<p align="center">
|
||||
<a href="./README.md">简体中文</a> |
|
||||
<a href="./README_ENG.md">English</a> |
|
||||
<a href="./README_JPN.md">日本語</a>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="https://dataelem.feishu.cn/wiki/ZxW6wZyAJicX4WkG0NqcWsbynde"><img src="https://img.shields.io/badge/docs-Wiki-brightgreen"></a>
|
||||
<img src="https://img.shields.io/github/license/dataelement/bisheng" alt="license"/>
|
||||
<img src="https://img.shields.io/docker/pulls/dataelement/bisheng-frontend" alt="docker-pull-count" />
|
||||
<a href=""><img src="https://img.shields.io/github/last-commit/dataelement/bisheng"></a>
|
||||
<a href="https://star-history.com/#dataelement/bisheng&Timeline"><img src="https://img.shields.io/github/stars/dataelement/bisheng?color=yellow"></a>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="./README_CN.md">简体中文</a> |
|
||||
<a href="./README.md">English</a> |
|
||||
<a href="./README_JPN.md">日本語</a>
|
||||
</p>
|
||||
|
||||
|
||||
<p align="center">
|
||||
<a href="https://trendshift.io/repositories/717" target="_blank"><img src="https://trendshift.io/api/badge/repositories/717" alt="dataelement%2Fbisheng | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
||||
</p>
|
||||
<div class="column" align="middle">
|
||||
<!-- <a href="https://bisheng.slack.com/join/shared_invite/"> -->
|
||||
<!-- <img src="https://img.shields.io/badge/Join-Slack-orange" alt="join-slack"/> -->
|
||||
@@ -22,137 +27,81 @@
|
||||
<!-- <img src="https://img.shields.io/docker/pulls/bisheng-io/bisheng" alt="docker-pull-count" /> -->
|
||||
</div>
|
||||
|
||||
# 欢迎来到 Bisheng
|
||||
|
||||
## Bisheng 是什么
|
||||
BISHENG is an open LLM application devops platform, focusing on enterprise scenarios. It has been used by a large number of industry leading organizations and Fortune 500 companies.
|
||||
|
||||
Bisheng是一款领先的开源<b>大模型应用开发平台</b>,赋能和加速大模型应用开发落地,帮助用户以最佳体验进入下一代应用开发模式。
|
||||
|
||||
“毕昇”是活字印刷术的发明人,活字印刷术为人类知识的传递起到了巨大的推动作用。我们希望“毕昇”同样能够为智能应用的广泛落地提供有力的支撑。欢迎大家一道参与。
|
||||
|
||||
Bisheng 基于 [Apache 2.0 License](https://github.com/dataelement/bisheng/blob/main/LICENSE) 协议发布,于 2023 年 8 月底正式开源。
|
||||
"Bi Sheng" was the inventor of movable type printing, which played a vital role in promoting the transmission of human knowledge. We hope that BISHENG can also provide strong support for the widespread implementation of intelligent applications. Everyone is welcome to participate.
|
||||
|
||||
|
||||
## 产品亮点
|
||||
## Features
|
||||
1. Unique [BISHENG Workflow](https://dataelem.feishu.cn/wiki/R7HZwH5ZGiJUDrkHZXicA9pInif)
|
||||
- 🧩 **Independent and comprehensive application orchestration framework**: Enables the execution of various tasks within a single framework (while similar products rely on bot invocation or separate chatflow and workflow modules for different tasks).
|
||||
- 🔄 **Human in the loop**: Allows users to intervene and provide feedback during the execution of workflows (including multi-turn conversations), whereas similar products can only execute workflows from start to finish without intervention.
|
||||
- 💥 **Powerful**: Supports loops, parallelism, batch processing, conditional logic, and free combination of all logic components. It also handles complex scenarios such as multi-type input/output, report generation, content review, and more.
|
||||
- 🖐️ **User-friendly and intuitive**: Operations like loops, parallelism, and batch processing, which require specialized components in similar products, can be easily visualized in BISHENG as a "flowchart" (drawing a loop forms a loop, aligning elements creates parallelism, and selecting multiple items enables batch processing).
|
||||
<p align="center"><img src="https://dataelem.com/bs/bisheng_workflow.png" alt="sence0"></p>
|
||||
|
||||
- 便捷:即使是业务人员,基于我们预置的应用模板,通过简单直观的表单填写方式快速搭建以大模型为核心的智能应用。
|
||||
- 灵活:对大模型技术有了解的人员,我们紧跟最前沿大模型技术生态提供数百种开发组件,基于可视化且自由的流程编排能力,可开发出任意类型的大模型应用,而不仅是简单的提示词工程。
|
||||
- 可靠与企业级:当前许多同类的开源项目仅适用于实验测试场景,缺少真正生产使用的企业级特性,包括:高并发下的高可用、应用运营及效果持续迭代优化、贴合真实业务场景的实用功能等,这些都是毕昇平台的差异化能力;另外,更直观的是,企业内的数据质量参差不齐,想要真正把所有数据利用起来,首先需要有完备的非结构化数据治理能力,而这是过去几年我们团队所积累的核心能力,在毕昇的demo环境中您可以通过相关组件直接接入这些能力,并且这些能力免费不限量使用。
|
||||
2. <b>Designed for Enterprise Applications</b>: Document review, fixed-layout report generation, multi-agent collaboration, policy update comparison, support ticket assistance, customer service assistance, meeting minutes generation, resume screening, call record analysis, unstructured data governance, knowledge mining, data analysis, and more.
|
||||
|
||||
The platform supports the construction of <b>highly complex enterprise application scenarios</b> and offers <b>deep optimization</b> with hundreds of components and thousands of parameters.
|
||||
<p align="center"><img src="https://dataelem.com/bs/chat.png" alt="sence1"></p>
|
||||
|
||||
## 产品应用
|
||||
3. <b>Enterprise-grade</b> features are the fundamental guarantee for application implementation: security review, RBAC, user group management, traffic control by group, SSO/LDAP, vulnerability scanning and patching, high availability deployment solutions, monitoring, statistics, and more.
|
||||
<p align="center"><img src="https://dataelem.com/bs/pro.png" alt="sence2"></p>
|
||||
|
||||
使用毕昇平台,我们可以搭建各类丰富的大模型应用:
|
||||
4. <b>High-Precision Document Parsing</b>: Our high-precision document parsing model is trained on a vast amount of high-quality data accumulated over past 5 years. It includes high-precision printed text, handwritten text, and rare character recognition models, table recognition models, layout analysis models, and seal models., table recognition models, layout analysis models, and seal models. You can deploy it privately for free.
|
||||
<p align="center"><img src="https://dataelem.com/bs/ocr.png" alt="sence3"></p>
|
||||
|
||||
分析报告生成
|
||||
5. A community for sharing best practices across various enterprise scenarios: An open repository of application cases and best practices.
|
||||
## Quick start
|
||||
|
||||
- 📃 合同审核报告生成
|
||||
- 🏦 信贷调查报告生成
|
||||
- 📈 招股书分析报告生成
|
||||
- 💼 智能投顾报告生成
|
||||
- 👀 文档摘要生成
|
||||
Please ensure the following conditions are met before installing BISHENG:
|
||||
- CPU >= 4 Virtual Cores
|
||||
- RAM >= 16 GB
|
||||
- Docker 19.03.9+
|
||||
- Docker Compose 1.25.1+
|
||||
> Recommended hardware condition: 18 virtual cores, 48G. In addition to installing BISHENG, we will also install the following third-party components by default: ES, Milvus, and Onlyoffice.
|
||||
|
||||
Download BISHENG
|
||||
```bash
|
||||
git clone https://github.com/dataelement/bisheng.git
|
||||
# Enter the installation directory
|
||||
cd bisheng/docker
|
||||
|
||||
知识库问答
|
||||
- 👩💻 用户手册问答
|
||||
- 👩🏻🔬 研报知识库问答
|
||||
- 🗄 规章制度问答
|
||||
- 💊 《中华药典》知识问答
|
||||
- 📊 股价数据库问答
|
||||
# If the system does not have the git command, you can download the BISHENG code as a zip file.
|
||||
wget https://github.com/dataelement/bisheng/archive/refs/heads/main.zip
|
||||
# Unzip and enter the installation directory
|
||||
unzip main.zip && cd bisheng-main/docker
|
||||
```
|
||||
Start BISHENG
|
||||
```bash
|
||||
docker compose -f docker-compose.yml -p bisheng up -d
|
||||
```
|
||||
After the startup is complete, access http://IP:3001 in the browser. The login page will appear, proceed with user registration.
|
||||
|
||||
By default, the first registered user will become the system admin.
|
||||
|
||||
对话
|
||||
- 🎭 扮演面试官对话
|
||||
- 📍 小红书文案助手
|
||||
- 👩🎤 扮演外教对话
|
||||
- 👨🏫 简历优化助手
|
||||
For more installation and deployment issues, refer to::[Self-hosting](https://dataelem.feishu.cn/wiki/BSCcwKd4Yiot3IkOEC8cxGW7nPc)
|
||||
|
||||
## Acknowledgement
|
||||
This repo benefits from [langchain](https://github.com/langchain-ai/langchain) [langflow](https://github.com/logspace-ai/langflow) [unstructured](https://github.com/Unstructured-IO/unstructured) and [LLaMA-Factory](https://github.com/hiyouga/LLaMA-Factory) . Thanks for their wonderful works.
|
||||
|
||||
要素提取
|
||||
<b>Thank you to our contributors:</b>
|
||||
|
||||
- 📄 合同关键要素提取
|
||||
- 🏗️ 工程报告要素提取
|
||||
- 🗂️ 通用元数据提取
|
||||
- 🎫 卡证票据要素提取
|
||||
|
||||
|
||||
各类应用构建方法详见:[应用案例](https://m7a7tqsztt.feishu.cn/wiki/ZfkmwLPfeiAhQSkK2WvcX87unxc)
|
||||
|
||||
我们认为在企业真实场景中,“对话”仅是众多交互形式中的一种,未来我们还将新增流程自动化、搜索等更多应用形态的支持。
|
||||
|
||||
|
||||
## 快速开始
|
||||
|
||||
### 启动 Bisheng
|
||||
|
||||
- [安装 Bisheng](https://m7a7tqsztt.feishu.cn/wiki/BSCcwKd4Yiot3IkOEC8cxGW7nPc)
|
||||
|
||||
|
||||
### 源码编译 Bisheng
|
||||
|
||||
- [编译Bisheng](https://dataelem.feishu.cn/wiki/EKdDw0IkyiNSAEkzc29cqKnmn7c)
|
||||
|
||||
获取更多内容,请阅读 [开发者文档](https://m7a7tqsztt.feishu.cn/wiki/ITmJwMXVliBnzpkW3nkcqPVrnse)。
|
||||
|
||||
|
||||
## 贡献代码
|
||||
|
||||
欢迎向 Bisheng 社区贡献你的代码。代码贡献流程或提交补丁等相关信息详见
|
||||
[代码贡献准则](https://github.com/dataelement/bisheng/blob/main/CONTRIBUTING.md)。
|
||||
参考 [社区仓库](https://github.com/dataelement/community) 了解社区管理准则并获取更多社区资源。
|
||||
|
||||
<!-- ### All contributors -->
|
||||
|
||||
<!-- Do not remove end of hero-bot -->
|
||||
<br>
|
||||
|
||||
### All Thanks To Our Contributors:
|
||||
<a href="https://github.com/dataelement/bisheng/graphs/contributors">
|
||||
<img src="https://contrib.rocks/image?repo=dataelement/bisheng" />
|
||||
</a>
|
||||
|
||||
## Bisheng 文档
|
||||
|
||||
获取更多有关安装、开发、部署和管理的指南,请查看 [Bisheng 文档](https://m7a7tqsztt.feishu.cn/wiki/ZxW6wZyAJicX4WkG0NqcWsbynde).
|
||||
|
||||
|
||||
## 社区
|
||||
|
||||
- 欢迎加入 [Slack](https://www.dataelem.com/) 频道分享你的建议与问题。
|
||||
- 你也可以通过 [FAQ](https://m7a7tqsztt.feishu.cn/wiki/XdGCwkDJviC0Z8klbdbcF790n9b) 页面,查看常见问题及解答。
|
||||
- 你也可以加入 [讨论组](https://github.com/dataelement/bisheng/discussions) 发起问题和讨论。
|
||||
|
||||
|
||||
<!-- 订阅 Bisheng 邮件:
|
||||
|
||||
- [Technical Steering Committee](https://www.dataelem.com/)
|
||||
- [Technical Discussions](https://www.dataelem.com/)
|
||||
- [Announcement](https://www.dataelem.com/) -->
|
||||
|
||||
关注 Bisheng 社交媒体:
|
||||
|
||||
<!-- - [知乎](https://www.zhihu.com/org/bisheng-io)
|
||||
- [CSDN](http://bishengio.blog.csdn.net/)
|
||||
- [Bilibili](http://space.bilibili.com/xxxxx) -->
|
||||
- Bisheng 技术交流微信群
|
||||
## Community & contact
|
||||
Welcome to join our discussion group
|
||||
|
||||
<img src="https://www.dataelem.com/nstatic/qrcode.png" alt="Wechat QR Code">
|
||||
|
||||
## 加入我们
|
||||
|
||||
DataElem Inc. 是 Bisheng 项目的幕后公司。我们正在 [招聘](https://www.dataelem.com/contact/team) 算法、开发和全栈工程师。欢迎加入我们,让我们携手构建下一代的智能应用开发平台。
|
||||
|
||||
|
||||
## 特别感谢
|
||||
|
||||
Bisheng 采用了以下依赖库:
|
||||
|
||||
- 感谢开源模型预估框架 [Triton](https://github.com/triton-inference-server) 。
|
||||
- 感谢开源LLM应用开发库 [langchain](https://github.com/langchain-ai/langchain)。
|
||||
- 感谢开源非结构化数据解析引擎 [unstructured](https://github.com/Unstructured-IO/unstructured)。
|
||||
- 感谢开源langchain可视化工具 [langflow](https://github.com/logspace-ai/langflow)。
|
||||
|
||||
|
||||
<!--
|
||||
## Star History
|
||||
|
||||
[](https://star-history.com/#dataelement/bisheng&Date)
|
||||
-->
|
||||
|
||||
+121
@@ -0,0 +1,121 @@
|
||||
<img src="https://dataelem.com/bs/face.png" alt="Bisheng banner">
|
||||
|
||||
<p align="center">
|
||||
<a href="https://dataelem.feishu.cn/wiki/ZxW6wZyAJicX4WkG0NqcWsbynde"><img src="https://img.shields.io/badge/docs-Wiki-brightgreen"></a>
|
||||
<img src="https://img.shields.io/github/license/dataelement/bisheng" alt="license"/>
|
||||
<img src="https://img.shields.io/docker/pulls/dataelement/bisheng-frontend" alt="docker-pull-count" />
|
||||
<a href=""><img src="https://img.shields.io/github/last-commit/dataelement/bisheng"></a>
|
||||
<a href="https://star-history.com/#dataelement/bisheng&Timeline"><img src="https://img.shields.io/github/stars/dataelement/bisheng?color=yellow"></a>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="./README_CN.md">简体中文</a> |
|
||||
<a href="./README.md">English</a> |
|
||||
<a href="./README_JPN.md">日本語</a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://trendshift.io/repositories/717" target="_blank"><img src="https://trendshift.io/api/badge/repositories/717" alt="dataelement%2Fbisheng | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
||||
</p>
|
||||
<div class="column" align="middle">
|
||||
<!-- <a href="https://bisheng.slack.com/join/shared_invite/"> -->
|
||||
<!-- <img src="https://img.shields.io/badge/Join-Slack-orange" alt="join-slack"/> -->
|
||||
</a>
|
||||
<!-- <img src="https://img.shields.io/github/license/bisheng-io/bisheng" alt="license"/> -->
|
||||
<!-- <img src="https://img.shields.io/docker/pulls/bisheng-io/bisheng" alt="docker-pull-count" /> -->
|
||||
</div>
|
||||
|
||||
|
||||
BISHENG毕昇 是一款 <b>开源</b> LLM应用开发平台,主攻<b>企业场景</b>, 已有大量行业头部组织及世界500强企业在使用。
|
||||
|
||||
“毕昇”是活字印刷术的发明人,活字印刷术为人类知识的传递起到了巨大的推动作用。我们希望“BISHENG毕昇”同样能够为智能应用的广泛落地提供有力支撑。欢迎大家一道参与。
|
||||
|
||||
|
||||
## 特点
|
||||
1. **独具特色的[BISHENG workflow](https://dataelem.feishu.cn/wiki/R7HZwH5ZGiJUDrkHZXicA9pInif)**
|
||||
|
||||
- 🧩 **独立、完备的应用编排框架**:可在一个框架下实现各类任务(同类产品需要被 bot 调用,或划分成 chatflow 与 workflow 来完成不同类型的任务)。
|
||||
- 🔄 **Human in the loop**:支持用户在Workflow执行的中间过程进行干预和反馈(包括多轮对话),而同类产品只能从头执行到尾。
|
||||
- 💥 **强大**:支持成环、并行、跑批、判断逻辑以及所有逻辑的任意自由组合;支持多类型输入输出、撰写报告、内容审核等复杂场景。
|
||||
- 🖐️ **易用、符合直觉**:如成环、并行、批量运行操作,在同类产品中用户需借助专门组件实现,在BISHENG中只需完全按照直觉连接成“流程图”即可(画圈成环、并列即并行、多选即批量)。
|
||||
<p align="center"><img src="https://dataelem.com/bs/bisheng_workflow.png" alt="sence0"></p>
|
||||
|
||||
2. **专为企业应用而生**:文档审核、固定版式报告生成、多智能体协作、规范制度更新差异比对、工单问答、客服辅助、会议纪要生成、简历筛选、通话记录分析、非结构化数据治理、知识挖掘、数据分析...平台支持高复杂度企业应用场景构建,支持数百个组件与数千个参数的深度调优。
|
||||
<p align="center"><img src="https://dataelem.com/bs/chat.png" alt="sence1"></p>
|
||||
|
||||
3. **企业级特性是应用落地的基本保障**:安全审查、基于角色的细颗粒度权限管理、用户组管理、分组流量控制、SSO/LDAP、漏洞扫描修复、高可用部署方案、监控、统计...
|
||||
<p align="center"><img src="https://dataelem.com/bs/pro.png" alt="sence2"></p>
|
||||
|
||||
4. **高精度文档解析**:5年海量数据沉淀,高精度文档解析模型支持免费私有化部署使用,包括高精度印刷体、手写体与生僻字识别模型、表格识别模型、版式分析模型、印章模型
|
||||
<p align="center"><img src="https://dataelem.com/bs/ocr.png" alt="sence3"></p>
|
||||
|
||||
5. **大量企业场景落地最佳实践分享社区**:开放的应用案例与最佳实践库。
|
||||
<p align="center"><img src="https://dataelem.com/bs/sence.png" alt="sence4"></p>
|
||||
|
||||
|
||||
## 快速安装
|
||||
|
||||
安装BISHENG前请先确保满足以下条件:
|
||||
- CPU >= 8 Core
|
||||
- RAM >= 32 GB
|
||||
- Docker 19.03.9+
|
||||
- Docker Compose 1.25.1+
|
||||
> 除了BISHENG前后端,我们默认还会安装第三方组件ES、Milvus、Onlyoffice
|
||||
|
||||
下载BISHENG代码
|
||||
```bash
|
||||
# 如果系统中有git命令,可以直接下载毕昇代码
|
||||
git clone https://github.com/dataelement/bisheng.git
|
||||
# 进入安装目录
|
||||
cd bisheng/docker
|
||||
|
||||
# 如果系统没有没有git命令,可以下载毕昇代码zip包
|
||||
wget https://github.com/dataelement/bisheng/archive/refs/heads/main.zip
|
||||
# 解压并进入安装目录
|
||||
unzip main.zip && cd bisheng-main/docker
|
||||
```
|
||||
启动BISHENG
|
||||
```bash
|
||||
# 进入bisheng/docker或bisheng-main/docker目录,执行
|
||||
docker compose -f docker-compose.yml -p bisheng up -d
|
||||
```
|
||||
启动后,在浏览器中访问 http://IP:3001 ,出现登录页,进行用户注册。默认第一个注册的用户会成为系统admin。
|
||||
|
||||
其他安装部署问题参考:[私有化部署](https://dataelem.feishu.cn/wiki/BSCcwKd4Yiot3IkOEC8cxGW7nPc)
|
||||
|
||||
|
||||
## 资源
|
||||
- [📄应用案例/场景库](https://dataelem.feishu.cn/wiki/ZfkmwLPfeiAhQSkK2WvcX87unxc)
|
||||
- [📄经验技巧](https://dataelem.feishu.cn/wiki/OWFRwknFaiIMajke4m5cFeLrnie)
|
||||
- [📄功能使用说明](https://dataelem.feishu.cn/wiki/WxH6wubbAiBkRIkSEyecmpDMnjF)
|
||||
- [📄BISHENG Blog](https://dataelem.feishu.cn/wiki/BiNowcaYWilewdksXQ5cZl3tnzy)
|
||||
|
||||
|
||||
## 感谢
|
||||
|
||||
感谢我们的贡献者:
|
||||
|
||||
<a href="https://github.com/dataelement/bisheng/graphs/contributors">
|
||||
<img src="https://contrib.rocks/image?repo=dataelement/bisheng" />
|
||||
</a>
|
||||
|
||||
|
||||
<br>
|
||||
Bisheng 采用了以下依赖库:
|
||||
|
||||
- 感谢开源LLM应用开发库 [langchain](https://github.com/langchain-ai/langchain)。
|
||||
- 感谢开源langchain可视化工具 [langflow](https://github.com/logspace-ai/langflow)。
|
||||
- 感谢开源非结构化数据解析引擎 [unstructured](https://github.com/Unstructured-IO/unstructured)。
|
||||
- 感谢开源LLM微调框架 [LLaMA-Factory](https://github.com/hiyouga/LLaMA-Factory) 。
|
||||
|
||||
|
||||
## 社区与支持
|
||||
欢迎加入我们的交流群
|
||||
|
||||
<img src="https://www.dataelem.com/nstatic/qrcode.png" alt="Wechat QR Code">
|
||||
|
||||
|
||||
<!--
|
||||
## Star History
|
||||
|
||||
[](https://star-history.com/#dataelement/bisheng&Date)
|
||||
-->
|
||||
-145
@@ -1,145 +0,0 @@
|
||||
<img src="https://www.dataelem.com/nstatic/bisheng.png" alt="Bisheng banner">
|
||||
|
||||
<p align="center">
|
||||
<a href="./README.md">简体中文</a> |
|
||||
<a href="./README_ENG.md">English</a> |
|
||||
<a href="./README_JPN.md">日本語</a>
|
||||
</p>
|
||||
|
||||
|
||||
<div class="column" align="middle">
|
||||
<!-- <a href="https://bisheng.slack.com/join/shared_invite/"> -->
|
||||
<!-- <img src="https://img.shields.io/badge/Join-Slack-orange" alt="join-slack"/> -->
|
||||
</a>
|
||||
<!-- <img src="https://img.shields.io/github/license/bisheng-io/bisheng" alt="license"/> -->
|
||||
<!-- <img src="https://img.shields.io/docker/pulls/bisheng-io/bisheng" alt="docker-pull-count" /> -->
|
||||
</div>
|
||||
|
||||
# Welcome to Bisheng
|
||||
|
||||
## What is Bisheng
|
||||
|
||||
Bisheng is a leading open-source platform for developing LLM applications. It empowers and accelerates the development of LLM applications and helps users to enter the next generation of application development mode with the best experience.
|
||||
|
||||
"Bisheng" is the inventor of movable type printing, which played a huge role in the dissemination of human knowledge. We hope that "Bisheng" can also provide strong support for the widespread landing of intelligent applications. Welcome to participate together.
|
||||
|
||||
Bisheng was released under the Apache 2.0 License at the end of August 2023.
|
||||
|
||||
|
||||
## Key Features
|
||||
|
||||
- Convenience: Even business person can quickly build intelligent applications centered around LLM through simple and intuitive form filling based on our pre-configured application templates.
|
||||
- Flexibility: For person familiar with LLM technologies, we provide hundreds of development components following the latest trends in the LLM technology ecosystem. With visual and flexible process orchestration capabilities, any type of LLM application can be developed, not just simple prompting projects.
|
||||
- Reliability and Enterprise-level: Many similar open-source projects are only suitable for experimental testing scenarios and lack enterprise-level features for real production use, including high availability under high concurrency, continuous iteration and optimization of application operations and effects, and practical functions that fit real business scenarios. These are the differentiated capabilities of the ByteDance platform. In addition, data quality within enterprises is uneven. To truly utilize all data, comprehensive unstructured data governance capabilities are needed, which is the core capability our team has accumulated over the past few years. In Bisheng's demo environment, you can directly access these capabilities through related components, and these capabilities are free and unlimited.
|
||||
|
||||
|
||||
## Product Applications
|
||||
|
||||
With the Bisheng platform, we can build a variety of LLM applications:
|
||||
|
||||
Analysis Report Generation:
|
||||
|
||||
- 📃 Contract Review Report Generation
|
||||
- 🏦 Credit Investigation Report Generation
|
||||
- 📈 IPO Analysis Report Generation
|
||||
- 💼 Intelligent Investment Advisory Report Generation
|
||||
- 👀 Document Summary Generation
|
||||
|
||||
|
||||
Knowledge Base Q&A:
|
||||
|
||||
- 👩💻 User Manual Q&A
|
||||
- 👩🔬 Research Report Knowledge Base Q&A
|
||||
- 🗄 Regulations and Rules Q&A
|
||||
- 💊 "Chinese Pharmacopoeia" Knowledge Q&A
|
||||
- 📊 Stock Price Database Q&A
|
||||
|
||||
|
||||
Dialogues:
|
||||
|
||||
- 🎭 Role-play as an interviewer
|
||||
- 📍 Xiaohongshu (Red Book) Copywriting Assistant
|
||||
- 👩🎤 Role-play as a foreign language teacher
|
||||
- 👨🏫 Resume Optimization Assistant
|
||||
|
||||
|
||||
Element Extraction:
|
||||
|
||||
- 📄 Key Elements Extraction from Contracts
|
||||
- 🏗️ Engineering Report Elements Extraction
|
||||
- 🗂️ General Metadata Extraction
|
||||
- 🎫 Key Elements Extraction from Cards and Bills
|
||||
|
||||
|
||||
For methods to build various applications, see[Application Cases](https://m7a7tqsztt.feishu.cn/wiki/ZfkmwLPfeiAhQSkK2WvcX87unxc).
|
||||
|
||||
We believe that in real enterprise scenarios, "dialogue" is just one of many interaction forms.
|
||||
In the future, we will also add support for more application forms such as process automation and search.
|
||||
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Start With Bisheng
|
||||
|
||||
- [Install Bisheng](https://m7a7tqsztt.feishu.cn/wiki/BSCcwKd4Yiot3IkOEC8cxGW7nPc)
|
||||
|
||||
|
||||
### Compile Bisheng From Src
|
||||
|
||||
Todo: update later
|
||||
|
||||
Get More Contents,Please Read [Dev Documents](https://m7a7tqsztt.feishu.cn/wiki/ITmJwMXVliBnzpkW3nkcqPVrnse)。
|
||||
|
||||
|
||||
## Contributing
|
||||
|
||||
Contributions to Bisheng are welcome from everyone. See [Guidelines for Contributing]((https://github.com/dataelement/bisheng/blob/main/CONTRIBUTING.md))
|
||||
for details on submitting patches and the contribution workflow.
|
||||
Refer [community repository](https://github.com/dataelement/community) to learn about our governance and access more community resources.
|
||||
|
||||
<!-- ### All contributors -->
|
||||
|
||||
<!-- Do not remove end of hero-bot -->
|
||||
<br>
|
||||
|
||||
## Bisheng Document
|
||||
|
||||
For more guides on installation, development, deployment, and management, please see [Bisheng Documentation](https://m7a7tqsztt.feishu.cn/wiki/ZxW6wZyAJicX4WkG0NqcWsbynde).
|
||||
|
||||
|
||||
## Community
|
||||
|
||||
- You're welcome to join our [Slack](https://www.dataelem.com/) channel to share your suggestions and issues.
|
||||
- You can also visit the [FAQ](https://m7a7tqsztt.feishu.cn/wiki/XdGCwkDJviC0Z8klbdbcF790n9b) page to see frequently asked questions and their answers.
|
||||
- You can also join the [Discussion Group](https://github.com/dataelement/bisheng/discussions) to raise questions and discussions.
|
||||
|
||||
|
||||
<!-- 订阅 Bisheng 邮件:
|
||||
|
||||
- [Technical Steering Committee](https://www.dataelem.com/)
|
||||
- [Technical Discussions](https://www.dataelem.com/)
|
||||
- [Announcement](https://www.dataelem.com/) -->
|
||||
|
||||
Follow Bisheng on social media:
|
||||
|
||||
<!-- - [知乎](https://www.zhihu.com/org/bisheng-io)
|
||||
- [CSDN](http://bishengio.blog.csdn.net/)
|
||||
- [Bilibili](http://space.bilibili.com/xxxxx) -->
|
||||
- Bisheng Technical Exchange WeChat Group
|
||||
|
||||
<img src="https://www.dataelem.com/nstatic/qrcode.png" alt="Wechat QR Code">
|
||||
|
||||
## Join Us
|
||||
|
||||
DataElem Inc. is the company behind the Bisheng project. We are [hiring](https://www.dataelem.com/contact/team) algorithm developers, developers, and full-stack engineers.
|
||||
Join us as we work together to build the next generation of intelligent application development platform
|
||||
|
||||
|
||||
## Acknowledgments
|
||||
|
||||
Bisheng adopts dependencies from the following:
|
||||
|
||||
- Thanks to the open-source model inference framework [Triton](https://github.com/triton-inference-server).
|
||||
- Thanks to the open-source LLM application development library [langchain](https://github.com/langchain-ai/langchain).
|
||||
- Thanks to the open-source unstructured data parsing engine [unstructured](https://github.com/Unstructured-IO/unstructured).
|
||||
- Thanks to the open-source langchain visualization tool [langflow](https://github.com/logspace-ai/langflow).
|
||||
+68
-107
@@ -1,12 +1,25 @@
|
||||
<img src="https://www.dataelem.com/nstatic/bisheng.png" alt="Bisheng banner">
|
||||
以下は、あなたが提供したMarkdownコンテンツの日本語翻訳です。
|
||||
|
||||
---
|
||||
|
||||
<img src="https://dataelem.com/bs/face.png" alt="Bisheng banner">
|
||||
|
||||
<p align="center">
|
||||
<a href="./README.md">简体中文</a> |
|
||||
<a href="./README_ENG.md">English</a> |
|
||||
<a href="https://dataelem.feishu.cn/wiki/ZxW6wZyAJicX4WkG0NqcWsbynde"><img src="https://img.shields.io/badge/docs-Wiki-brightgreen"></a>
|
||||
<img src="https://img.shields.io/github/license/dataelement/bisheng" alt="license"/>
|
||||
<img src="https://img.shields.io/docker/pulls/dataelement/bisheng-frontend" alt="docker-pull-count" />
|
||||
<a href=""><img src="https://img.shields.io/github/last-commit/dataelement/bisheng"></a>
|
||||
<a href="https://star-history.com/#dataelement/bisheng&Timeline"><img src="https://img.shields.io/github/stars/dataelement/bisheng?color=yellow"></a>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="./README_CN.md">简体中文</a> |
|
||||
<a href="./README.md">English</a> |
|
||||
<a href="./README_JPN.md">日本語</a>
|
||||
</p>
|
||||
|
||||
|
||||
<p align="center">
|
||||
<a href="https://trendshift.io/repositories/717" target="_blank"><img src="https://trendshift.io/api/badge/repositories/717" alt="dataelement%2Fbisheng | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
||||
</p>
|
||||
<div class="column" align="middle">
|
||||
<!-- <a href="https://bisheng.slack.com/join/shared_invite/"> -->
|
||||
<!-- <img src="https://img.shields.io/badge/Join-Slack-orange" alt="join-slack"/> -->
|
||||
@@ -15,131 +28,79 @@
|
||||
<!-- <img src="https://img.shields.io/docker/pulls/bisheng-io/bisheng" alt="docker-pull-count" /> -->
|
||||
</div>
|
||||
|
||||
# Bisheng へようこそ
|
||||
BISHENGは、エンタープライズシナリオに焦点を当てたオープンなLLMアプリケーションDevOpsプラットフォームです。多くの業界リーディング企業やフォーチュン500企業で使用されています。
|
||||
|
||||
## Bisheng とは
|
||||
「畢昇(Bi Sheng)」は、活版印刷の発明者であり、人類の知識の伝播に重要な役割を果たしました。我々は、BISHENGがインテリジェントアプリケーションの広範な実装に強力なサポートを提供できることを願っています。皆さんの参加を歓迎します。
|
||||
|
||||
Bisheng は、LLM アプリケーション開発のための主要なオープンソースプラットフォームです。Bisheng は、LLM アプリケーションの開発を強化し、加速し、ユーザーが最高の経験を持つ次世代のアプリケーション開発モードに入るのを支援します。
|
||||
## 特徴
|
||||
1. **独自の特徴を持つ[BISHENG workflow](https://dataelem.feishu.cn/wiki/R7HZwH5ZGiJUDrkHZXicA9pInif)**
|
||||
|
||||
- 🧩 **独立性と完備性を備えたアプリケーションオーケストレーションフレームワーク**:1つのフレームワーク内でさまざまなタスクを実現可能(類似製品では、botの呼び出しが必要だったり、chatflowとworkflowに分けて異なるタスクを処理する必要があります)。
|
||||
- 🔄 **Human in the loop**:Workflowの実行途中でユーザーが介入やフィードバック(多ターン対話を含む)を行えます(類似製品では最初から最後まで一貫して実行されるのみ)。
|
||||
- 💥 **強力な機能**:ループ化、並列処理、一括処理、条件分岐ロジック、さらにこれら全ての自由な組み合わせが可能です。多種類の入出力、レポート作成、コンテンツ審査といった複雑なシナリオも対応可能。
|
||||
- 🖐️ **直感的で使いやすい**:類似製品では専用のコンポーネントを使用する必要があるループ化、並列処理、一括処理操作も、BISHENGでは直感的に「フローチャート」として接続するだけで実現可能です(円を描けばループ化、並列に配置すれば並列処理、複数選択すれば一括処理)。
|
||||
|
||||
"Bisheng" は可動活字印刷の発明者であり、人類の知識の普及に大きな役割を果たしました。"Bisheng" もまた、インテリジェントアプリケーションの広範な着陸のための強力なサポートを提供することができることを願っています。一緒に参加しましょう。
|
||||
<p align="center"><img src="https://dataelem.com/bs/bisheng_workflow.png" alt="sence0"></p>
|
||||
|
||||
2. **エンタープライズアプリケーション向けに設計**: ドキュメントレビュー、固定レイアウトレポート生成、マルチエージェント協働、ポリシー更新比較、サポートチケット支援、カスタマーサービス支援、会議議事録生成、履歴書スクリーニング、通話記録分析、非構造化データガバナンス、知識採掘、データ分析など。プラットフォームは、**高度に複雑なエンタープライズアプリケーションシナリオの構築**をサポートし、**深い最適化**を行い、数百のコンポーネントと数千のパラメータを提供します。
|
||||
<p align="center"><img src="https://dataelem.com/bs/chat.png" alt="sence1"></p>
|
||||
|
||||
Bishengは2023年8月末にApache 2.0 Licenseの下でリリースされた。
|
||||
3. **エンタープライズグレード**の機能は、アプリケーション実装の基本的な保証です: セキュリティレビュー、RBAC、ユーザーグループ管理、グループごとのトラフィックコントロール、SSO/LDAP、脆弱性スキャンとパッチ適用、高可用性デプロイメントソリューション、モニタリング、統計など。
|
||||
<p align="center"><img src="https://dataelem.com/bs/pro.png" alt="sence2"></p>
|
||||
|
||||
4. **高精度ドキュメント解析**: 私たちの高精度ドキュメント解析モデルは、過去5年間にわたる大量の高品質データに基づいてトレーニングされています。高精度な印刷テキスト、手書きテキスト、稀少文字認識モデル、テーブル認識モデル、レイアウト解析モデル、印鑑モデルを含みます。プライベートに無料で展開することができます。
|
||||
<p align="center"><img src="https://dataelem.com/bs/ocr.png" alt="sence3"></p>
|
||||
|
||||
## 主な特徴
|
||||
|
||||
- 利便性: ビジネスパーソンでも、LLM を中心としたインテリジェントなアプリケーションを、あらかじめ設定されたアプリケーションテンプレートに基づき、シンプルで直感的なフォーム入力によって素早く構築することができます。
|
||||
- 柔軟性: LLM テクノロジーに精通した方には、LLM テクノロジーエコシステムの最新トレンドに沿った数百の開発コンポーネントを提供しています。視覚的で柔軟なプロセスオーケストレーション機能により、単純なプロンプトプロジェクトだけでなく、あらゆるタイプの LLM アプリケーションを開発することができます。
|
||||
- 信頼性とエンタープライズレベル: 同様のオープンソースプロジェクトの多くは、実験的なテストシナリオにしか適しておらず、高同時実行下での高可用性、アプリケーションの操作と効果の継続的な反復と最適化、実際のビジネスシナリオに適合する実用的な機能など、実際の本番環境で使用するためのエンタープライズレベルの機能が欠けています。これらは ByteDance プラットフォームの差別化された機能である。さらに、企業内のデータ品質にはばらつきがある。すべてのデータを真に活用するためには、包括的な非構造化データガバナンス能力が必要であり、これこそが、私たちのチームが過去数年にわたって蓄積してきた中核的機能なのです。Bisheng のデモ環境では、関連コンポーネントを通じてこれらの機能に直接アクセスすることができ、これらの機能は無料で無制限です。
|
||||
|
||||
|
||||
## 製品アプリケーション
|
||||
|
||||
Bishengプラットフォームでは、様々なLLMアプリケーションを構築することができます:
|
||||
|
||||
分析レポート生成:
|
||||
|
||||
- 📃 契約審査レポート生成
|
||||
- 🏦 信用調査レポート生成
|
||||
- 📈 IPO 分析レポート生成
|
||||
- 💼 インテリジェント投資アドバイザリーレポート生成
|
||||
- 👀 文書要約生成
|
||||
|
||||
|
||||
ナレッジベース Q&A:
|
||||
|
||||
- 👩💻 ユーザーマニュアル Q&A
|
||||
- 👩🔬 調査報告書ナレッジベース Q&A
|
||||
- 🗄 法規と規則 Q&A
|
||||
- 💊 「中国薬局方」知識 Q&A
|
||||
- 📊 株価データベース Q&A
|
||||
|
||||
|
||||
対話:
|
||||
|
||||
- 🎭 面接官のロールプレイ
|
||||
- 📍 小本集(赤本)コピーライティングアシスタント
|
||||
- 👩🎤 外国語教師のロールプレイ
|
||||
- 👨🏫 履歴書最適化アシスタント
|
||||
|
||||
|
||||
要素の抽出:
|
||||
|
||||
- 📄 契約書からの主要要素の抽出
|
||||
- 🏗️ エンジニアリングレポート要素抽出
|
||||
- 🗂️ 一般的なメタデータの抽出
|
||||
- 🎫 カードと請求書からのキーエレメントの抽出
|
||||
|
||||
|
||||
様々なアプリケーションを構築する方法については、[アプリケーションケース](https://m7a7tqsztt.feishu.cn/wiki/ZfkmwLPfeiAhQSkK2WvcX87unxc)を参照してください。
|
||||
|
||||
私たちは、実際の企業シナリオにおいて、「対話」は数ある対話形式のひとつに過ぎないと考えています。
|
||||
将来的には、プロセスの自動化や検索など、より多くのアプリケーションのサポートも追加していく予定です。
|
||||
5. 様々なエンタープライズシナリオにおけるベストプラクティスを共有するコミュニティ: オープンなアプリケーションケースとベストプラクティスのリポジトリ。
|
||||
|
||||
|
||||
## クイックスタート
|
||||
|
||||
### Bisheng を始める
|
||||
BISHENGをインストールする前に、以下の条件を満たしていることを確認してください:
|
||||
- CPU >= 8 コア
|
||||
- RAM >= 32 GB
|
||||
- Docker 19.03.9以上
|
||||
- Docker Compose 1.25.1以上
|
||||
|
||||
- [Bisheng のインストール](https://m7a7tqsztt.feishu.cn/wiki/BSCcwKd4Yiot3IkOEC8cxGW7nPc)
|
||||
> BISHENGをインストールする際、デフォルトで以下のサードパーティコンポーネントもインストールされます: ES, Milvus, Onlyoffice。
|
||||
|
||||
BISHENGのダウンロード
|
||||
```bash
|
||||
git clone https://github.com/dataelement/bisheng.git
|
||||
# インストールディレクトリに移動
|
||||
cd bisheng/docker
|
||||
|
||||
### ソースからの Bisheng のコンパイル
|
||||
# システムにgitコマンドがない場合は、BISHENGのコードをzipファイルとしてダウンロードできます。
|
||||
wget https://github.com/dataelement/bisheng/archive/refs/heads/main.zip
|
||||
# 解凍してインストールディレクトリに移動
|
||||
unzip main.zip && cd bisheng-main/docker
|
||||
```
|
||||
|
||||
- [Bisheng のコンパイル](https://dataelem.feishu.cn/wiki/EKdDw0IkyiNSAEkzc29cqKnmn7c)
|
||||
BISHENGの起動
|
||||
```bash
|
||||
docker compose -f docker-compose.yml -p bisheng up -d
|
||||
```
|
||||
|
||||
より多くのコンテンツを入手するには、[開発ドキュメント](https://m7a7tqsztt.feishu.cn/wiki/ITmJwMXVliBnzpkW3nkcqPVrnse)をお読みください。
|
||||
起動完了後、ブラウザでhttp://IP:3001にアクセスします。ログインページが表示されるので、ユーザー登録を行います。
|
||||
|
||||
デフォルトでは、最初に登録されたユーザーがシステム管理者となります。
|
||||
|
||||
## コントリビュート
|
||||
詳細なインストールおよびデプロイに関する問題は、こちらを参照してください:[私有化部署](https://dataelem.feishu.cn/wiki/BSCcwKd4Yiot3IkOEC8cxGW7nPc)
|
||||
|
||||
Bisheng へのコントリビュートは、どなたでも歓迎いたします。
|
||||
パッチの送信とコントリビュートのワークフローの詳細については、[Guidelines for Contributing]((https://github.com/dataelement/bisheng/blob/main/CONTRIBUTING.md)) を参照してください。
|
||||
[コミュニティリポジトリ](https://github.com/dataelement/community)を参照し、私たちのガバナンスについて学び、より多くのコミュニティリソースにアクセスしてください。
|
||||
## 謝辞
|
||||
このリポジトリは [langchain](https://github.com/langchain-ai/langchain) [langflow](https://github.com/logspace-ai/langflow) [unstructured](https://github.com/Unstructured-IO/unstructured) および [LLaMA-Factory](https://github.com/hiyouga/LLaMA-Factory) の恩恵を受けています。素晴らしい作品に感謝します。
|
||||
|
||||
<!-- ### All contributors -->
|
||||
**貢献者に感謝します:**
|
||||
|
||||
<!-- Do not remove end of hero-bot -->
|
||||
<br>
|
||||
<a href="https://github.com/dataelement/bisheng/graphs/contributors">
|
||||
<img src="https://contrib.rocks/image?repo=dataelement/bisheng" />
|
||||
</a>
|
||||
|
||||
## Bisheng ドキュメント
|
||||
|
||||
インストール、開発、デプロイ、管理に関する詳しいガイドは、[Bisheng Documentation](https://m7a7tqsztt.feishu.cn/wiki/ZxW6wZyAJicX4WkG0NqcWsbynde) を参照。
|
||||
|
||||
|
||||
## コミュニティ
|
||||
|
||||
- 私たちの [Slack](https://www.dataelem.com/) チャンネルに参加して、提案や問題を共有してください。
|
||||
- また、[FAQ](https://m7a7tqsztt.feishu.cn/wiki/XdGCwkDJviC0Z8klbdbcF790n9b) のページでは、よくある質問とその回答をご覧いただけます。
|
||||
- また、[ディスカッショングループ](https://github.com/dataelement/bisheng/discussions)に参加して質問やディスカッションをすることもできます。
|
||||
|
||||
|
||||
<!-- Bisheng のメールを購読する:
|
||||
|
||||
- [Technical Steering Committee](https://www.dataelem.com/)
|
||||
- [Technical Discussions](https://www.dataelem.com/)
|
||||
- [Announcement](https://www.dataelem.com/) -->
|
||||
|
||||
Bisheng をソーシャルメディアでフォローする:
|
||||
|
||||
<!-- - [知乎](https://www.zhihu.com/org/bisheng-io)
|
||||
- [CSDN](http://bishengio.blog.csdn.net/)
|
||||
- [Bilibili](http://space.bilibili.com/xxxxx) -->
|
||||
- Bisheng 技術交流 WeChat グループ
|
||||
## コミュニティと連絡先
|
||||
ディスカッショングループへの参加を歓迎します。
|
||||
|
||||
<img src="https://www.dataelem.com/nstatic/qrcode.png" alt="Wechat QR Code">
|
||||
|
||||
## 参加しましょう
|
||||
---
|
||||
|
||||
DataElem Inc. は、Bisheng プロジェクトの運営会社です。アルゴリズム開発者、開発者、フルスタックエンジニアを募集しています。
|
||||
次世代インテリジェントアプリケーション開発プラットフォームの構築に向け、共に取り組みましょう。
|
||||
|
||||
|
||||
## 謝辞
|
||||
|
||||
Bisheng は以下のライブラリを使用しています:
|
||||
|
||||
- オープンソースのモデル推論フレームワーク [Triton](https://github.com/triton-inference-server) に感謝します。
|
||||
- オープンソースの LLM アプリケーション開発ライブラリ [LangChain](https://github.com/langchain-ai/langchain) に感謝します。
|
||||
- オープンソースの非構造化データ解析エンジン [unstructured](https://github.com/Unstructured-IO/unstructured) に感謝します。
|
||||
- オープンソースの LangChain 可視化ツール [langflow](https://github.com/logspace-ai/langflow) に感謝します。
|
||||
この翻訳を使用して、Markdownファイルを作成できます。
|
||||
|
||||
+39
@@ -0,0 +1,39 @@
|
||||
# Security Policy
|
||||
|
||||
## Reporting Security Issues
|
||||
|
||||
We take the security of our project seriously. If you believe you have found a security vulnerability, please report it to us privately. **Please do not report security vulnerabilities through public GitHub issues, discussions, or pull requests.**
|
||||
|
||||
> **Important Note**: Any code within the `classic/` folder is considered legacy, unsupported, and out of scope for security reports. We will not address security vulnerabilities in this deprecated code.
|
||||
|
||||
Instead, please report them via:
|
||||
- [GitHub Security Advisory](https://github.com/dataelement/bisheng/security/advisories/new)
|
||||
<!--- [Huntr.dev](https://huntr.com/repos/significant-gravitas/autogpt) - where you may be eligible for a bounty-->
|
||||
|
||||
### Reporting Process
|
||||
1. **Submit Report**: Use one of the above channels to submit your report
|
||||
2. **Response Time**: Our team will acknowledge receipt of your report within 14 business days.
|
||||
3. **Collaboration**: We will collaborate with you to understand and validate the issue
|
||||
4. **Resolution**: We will work on a fix and coordinate the release process
|
||||
|
||||
|
||||
### Disclosure Policy
|
||||
- Please provide detailed reports with reproducible steps
|
||||
- Include the version/commit hash where you discovered the vulnerability
|
||||
- Allow us a 90-day security fix window before any public disclosure
|
||||
- Share any potential mitigations or workarounds if known
|
||||
|
||||
## Supported Versions
|
||||
Only the following versions are eligible for security updates:
|
||||
|
||||
| Version | Supported |
|
||||
|---------|-----------|
|
||||
| Latest release on master branch | ✅ |
|
||||
| Development commits (pre-master) | ✅ |
|
||||
| Classic folder (deprecated) | ❌ |
|
||||
| All other versions | ❌ |
|
||||
|
||||
|
||||
|
||||
---
|
||||
Last updated: November 2024
|
||||
@@ -32,7 +32,8 @@ pdf_model_params:
|
||||
table_model_ep: "http://192.168.106.12:9001/v2.1/models/elem_table_detect_v1/infer"
|
||||
ocr_model_ep: "http://192.168.106.12:9001/v2.1/models/elem_ocr_collection_v3/infer"
|
||||
|
||||
|
||||
# 是否全部走ocr识别, false的话则由代码逻辑判断是否需要走ocr识别
|
||||
is_all_ocr: false
|
||||
# ocr识别需要的配置项
|
||||
ocr_conf:
|
||||
params:
|
||||
|
||||
@@ -20,6 +20,40 @@ redis_url: "redis://redis:6379/1"
|
||||
# sentinel_password: encrypt(gAAAAABlp4b4c59FeVGF_OQRVf6NOUIGdxq8246EBD-b0hdK_jVKRs1x4PoAn0A6C5S6IiFKmWn0Nm5eBUWu-7jxcqw6TiVjQA==)
|
||||
# db: 1
|
||||
|
||||
# celery的broken地址
|
||||
celery_redis_url: "redis://redis:6379/2"
|
||||
celery_task:
|
||||
# 对celery熟悉的用户可以自定义配置任务的路由,启动不同类型的worker处理不同类型的异步任务。注意工作流的执行只能在一个进程内!!!
|
||||
task_routers:
|
||||
bisheng.worker.knowledge.*: # 知识库文件处理相关任务
|
||||
queue: knowledge_celery
|
||||
bisheng.worker.workflow.*: # 工作流相关任务
|
||||
queue: workflow_celery
|
||||
|
||||
# 知识库的milvus和es配置 支持使用 !env ${PATH} 填写环境变量的值, 若环境变量不存在则会报错
|
||||
vector_stores:
|
||||
milvus:
|
||||
connection_args: !env ${BS_MILVUS_CONNECTION_ARGS}
|
||||
is_partition: !env ${BS_MILVUS_IS_PARTITION}
|
||||
partition_suffix: !env ${BS_MILVUS_PARTITION_SUFFIX}
|
||||
elasticsearch:
|
||||
url: !env ${BS_ELASTICSEARCH_URL}
|
||||
ssl_verify: !env ${BS_ELASTICSEARCH_SSL_VERIFY}
|
||||
|
||||
|
||||
# 对象存储, 目前只支持minio
|
||||
object_storage:
|
||||
type: minio
|
||||
minio:
|
||||
schema: !env ${BS_MINIO_SCHEMA}
|
||||
cert_check: !env ${BS_MINIO_CERT_CHECK}
|
||||
endpoint: !env ${BS_MINIO_ENDPOINT}
|
||||
sharepoint: !env ${BS_MINIO_SHAREPOINT}
|
||||
access_key: !env ${BS_MINIO_ACCESS_KEY}
|
||||
secret_key: !env ${BS_MINIO_SECRET_KEY}
|
||||
public_bucket: 'bisheng' # 公共bucket,存储平台上一些需要持久化的文件。会设置为可公开访问
|
||||
tmp_bucket: 'tmp-dir' # 临时bucket,会对传到此bucket内的文件设置有效期
|
||||
|
||||
environment:
|
||||
env: dev
|
||||
uns_support: ['png','jpg','jpeg','bmp','doc', 'docx', 'ppt', 'pptx', 'xls', 'xlsx', 'txt', 'md', 'html', 'pdf', 'csv', 'tiff']
|
||||
@@ -37,19 +71,11 @@ logger_conf:
|
||||
# 日志级别
|
||||
level: INFO
|
||||
# 日志格式化函数,extra内支持trace_id
|
||||
format: "[{time:YYYY-MM-DD HH:mm:ss.SSSSSS}]|{level}|BISHENG|{extra[trace_id]}|{process.id}|{thread.id}|{message}"
|
||||
format: '<level>[{time:YYYY-MM-DD HH:mm:ss.SSSSSS}] [{level.name} process-{process.id}-{thread.id} {name}:{line}]</level> - <level>trace={extra[trace_id]} {message}</level>'
|
||||
# 每天的几点进行切割
|
||||
rotation: "00:00"
|
||||
retention: "3 Days"
|
||||
enqueue: ture
|
||||
- sink: "/app/data/err-v0-BISHENG-{HOSTNAME}.log"
|
||||
level: ERROR
|
||||
# 和原生不一样,后端会将配置使用eval()执行转为函数用来过滤特定日志级别。推荐lambda
|
||||
filter: "lambda record: record['level'].name == 'ERROR'"
|
||||
format: "[{time:YYYY-MM-DD HH:mm:ss.SSSSSS}]|{level}|BISHENG|{extra[trace_id]}||{process.id}|{thread.id}|||#EX_ERR:POS={name},line {line},ERR=500,EMSG={message}"
|
||||
rotation: "00:00"
|
||||
retention: "3 Days"
|
||||
enqueue: ture
|
||||
- sink: "/app/data/statistic.log"
|
||||
level: INFO
|
||||
# 和原生不一样,后端会将配置使用eval()执行转为函数用来过滤特定日志级别。推荐lambda
|
||||
@@ -57,4 +83,4 @@ logger_conf:
|
||||
format: "[{time:YYYY-MM-DD HH:mm:ss.SSSSSS}]|{level}|BISHENG|{extra[trace_id]}||{process.id}|{thread.id}|||#EX_ERR:POS={name},line {line},ERR=500,EMSG={message}"
|
||||
rotation: "00:00"
|
||||
retention: "3 Days"
|
||||
enqueue: ture
|
||||
enqueue: ture
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
#!/bin/bash
|
||||
|
||||
start_mode=${1:-api}
|
||||
|
||||
if [ $start_mode = "api" ]; then
|
||||
echo "Starting API server..."
|
||||
uvicorn bisheng.main:app --host 0.0.0.0 --port 7860 --no-access-log --workers 8
|
||||
elif [ $start_mode = "worker" ]; then
|
||||
echo "Starting Celery worker..."
|
||||
# 处理知识库相关任务的worker
|
||||
nohup celery -A bisheng.worker.main worker -l info -c 20 -P threads -Q knowledge_celery &
|
||||
# 工作流执行worker,只能启动一个进程来处理工作流的执行,暂不支持多进程
|
||||
celery -A bisheng.worker.main worker -l info -c 100 -P threads -Q workflow_celery
|
||||
else
|
||||
echo "Invalid start mode. Use 'api' or 'celery'."
|
||||
exit 1
|
||||
fi
|
||||
@@ -0,0 +1,29 @@
|
||||
services:
|
||||
ft_server:
|
||||
container_name: bisheng-ft-server
|
||||
image: dataelement/bisheng-ft:v0.2.0
|
||||
ports:
|
||||
- "8000:8000"
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng-ft/config.yaml:/opt/bisheng-ft/sft_server/config.yaml # 服务启动所需的配置文件地址,默认不用改
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/data/llm:/opt/bisheng-ft/models/model_repository # 这个是存放基座模型的目录,挂载到本机目录
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/data/finetune_output:/opt/bisheng-ft/finetune_output # 这个是存放微调过程的中间日志和微调训练后模型的目录,挂载到本机目录,不能与存放基座模型的目录相同
|
||||
security_opt:
|
||||
- seccomp:unconfined
|
||||
command: bash start-sft-server.sh # 启动服务
|
||||
restart: on-failure
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
||||
start_period: 30s
|
||||
interval: 90s
|
||||
timeout: 30s
|
||||
retries: 3
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
@@ -1,76 +0,0 @@
|
||||
services:
|
||||
bisheng-rt:
|
||||
container_name: bisheng-rt
|
||||
image: dataelement/bisheng-rt:0.0.6.3rc1
|
||||
shm_size: 10gb
|
||||
ports:
|
||||
- "9000:9000"
|
||||
- "9001:9001"
|
||||
- "9002:9002"
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- capabilities: [gpu]
|
||||
driver: nvidia
|
||||
device_ids: ['0,1'] # 指定想映射给rt服务使用的宿主机上的GPU ID号,如想映射多个卡,可写为['0','1','2']
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
# 不使用闭源模型的话,用下面的启动命令
|
||||
command: ["./bin/rtserver", "f"]
|
||||
# 使用闭源模型的话,用下面的启动命令,地址替换为授权地址
|
||||
# command: ["bash", "bin/entrypoint.sh", "--serveraddr=<license srv host>"]
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/data/llm:/opt/bisheng-rt/models/model_repository # 冒号前为宿主机上放置模型目录的路径,请根据实际环境修改;冒号后为映射到容器内的路径,请勿修改
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:9001/v2"]
|
||||
interval: 30s
|
||||
timeout: 20s
|
||||
retries: 3
|
||||
restart: on-failure
|
||||
|
||||
ft_server:
|
||||
container_name: bisheng-ft-server
|
||||
image: dataelement/bisheng-ft:latest
|
||||
ports:
|
||||
- "8000:8000"
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng-ft/config.yaml:/opt/bisheng-ft/sft_server/config.yaml # 服务启动所需的配置文件地址
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/data/llm:/opt/bisheng-ft/models/model_repository # 配置和RT服务同样的大模型目录
|
||||
security_opt:
|
||||
- seccomp:unconfined
|
||||
command: bash start-sft-server.sh # 启动服务
|
||||
restart: on-failure
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
||||
start_period: 30s
|
||||
interval: 90s
|
||||
timeout: 30s
|
||||
retries: 3
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
|
||||
bisheng-unstructured:
|
||||
container_name: bisheng-unstructured
|
||||
image: dataelement/bisheng-unstructured:latest
|
||||
ports:
|
||||
- "10001:10001"
|
||||
environment:
|
||||
rt_server: bisheng-rt:9001
|
||||
TZ: Asia/Shanghai
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng-uns/config.yaml:/opt/bisheng-unstructured/bisheng_unstructured/config/config.yaml
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:10001/health"]
|
||||
interval: 30s
|
||||
timeout: 20s
|
||||
retries: 3
|
||||
restart: on-failure
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
services:
|
||||
bisheng-unstructured:
|
||||
container_name: bisheng-unstructured
|
||||
image: dataelement/bisheng-unstructured:v0.0.3.14
|
||||
ports:
|
||||
- "10001:10001"
|
||||
environment:
|
||||
# 填写ocr_sdk或rt服务的根地址
|
||||
# server_address: bisheng-rt:9001
|
||||
# 这里填 ocr_sdk 或 rt
|
||||
# server_type: ocr_sdk
|
||||
TZ: Asia/Shanghai
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng-uns/config.yaml:/opt/bisheng-unstructured/bisheng_unstructured/config/config.yaml
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:10001/health"]
|
||||
interval: 30s
|
||||
timeout: 20s
|
||||
retries: 3
|
||||
restart: on-failure
|
||||
|
||||
@@ -40,12 +40,12 @@ services:
|
||||
|
||||
office:
|
||||
container_name: bisheng-office
|
||||
image: onlyoffice/documentserver:7.2.1
|
||||
image: onlyoffice/documentserver:7.1.1
|
||||
ports:
|
||||
- "8701:80"
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
JWT_ENABLED: false
|
||||
JWT_ENABLED: "false"
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/office/bisheng:/var/www/onlyoffice/documentserver/sdkjs-plugins/bisheng
|
||||
command: bash -c "supervisorctl restart all"
|
||||
@@ -53,17 +53,29 @@ services:
|
||||
|
||||
backend:
|
||||
container_name: bisheng-backend
|
||||
image: dataelement/bisheng-backend:latest
|
||||
image: dataelement/bisheng-backend:v1.3.1
|
||||
ports:
|
||||
- "7860:7860"
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
BS_MILVUS_CONNECTION_ARGS: '{"host":"milvus","port":"19530","user":"","password":"","secure":false}'
|
||||
BS_MILVUS_IS_PARTITION: 'true'
|
||||
BS_MILVUS_PARTITION_SUFFIX: '1'
|
||||
BS_ELASTICSEARCH_URL: 'http://elasticsearch:9200'
|
||||
BS_ELASTICSEARCH_SSL_VERIFY: '{}' # 可根据自己部署的密码进行配置 '{"basic_auth": ("elastic", "elastic")}'
|
||||
BS_MINIO_SCHEMA: 'false'
|
||||
BS_MINIO_CERT_CHECK: 'false'
|
||||
BS_MINIO_ENDPOINT: 'minio:9000'
|
||||
BS_MINIO_SHAREPOINT: 'minio:9000'
|
||||
BS_MINIO_ACCESS_KEY: 'minioadmin'
|
||||
BS_MINIO_SECRET_KEY: 'minioadmin'
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng/config/config.yaml:/app/bisheng/config.yaml
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng/entrypoint.sh:/app/entrypoint.sh
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/data/bisheng:/app/data
|
||||
security_opt:
|
||||
- seccomp:unconfined
|
||||
command: bash -c "uvicorn bisheng.main:app --host 0.0.0.0 --port 7860 --no-access-log --workers 2" # --workers 表示使用几个进程,提高并发度
|
||||
command: sh entrypoint.sh api # 启动api服务
|
||||
restart: on-failure
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:7860/health"]
|
||||
@@ -78,10 +90,42 @@ services:
|
||||
condition: service_healthy
|
||||
office:
|
||||
condition: service_started
|
||||
|
||||
backend_worker:
|
||||
container_name: bisheng-backend-worker
|
||||
image: dataelement/bisheng-backend:v1.3.1
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
BS_MILVUS_CONNECTION_ARGS: '{"host":"milvus","port":"19530","user":"","password":"","secure":false}'
|
||||
BS_MILVUS_IS_PARTITION: 'true'
|
||||
BS_MILVUS_PARTITION_SUFFIX: '1'
|
||||
BS_ELASTICSEARCH_URL: 'http://elasticsearch:9200'
|
||||
BS_ELASTICSEARCH_SSL_VERIFY: '{}' # 可根据自己部署的密码进行配置 '{"basic_auth": ("elastic", "elastic")}'
|
||||
BS_MINIO_SCHEMA: 'false'
|
||||
BS_MINIO_CERT_CHECK: 'false'
|
||||
BS_MINIO_ENDPOINT: 'minio:9000'
|
||||
BS_MINIO_SHAREPOINT: 'minio:9000'
|
||||
BS_MINIO_ACCESS_KEY: 'minioadmin'
|
||||
BS_MINIO_SECRET_KEY: 'minioadmin'
|
||||
volumes:
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng/config/config.yaml:/app/bisheng/config.yaml
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/bisheng/entrypoint.sh:/app/entrypoint.sh
|
||||
- ${DOCKER_VOLUME_DIRECTORY:-.}/data/bisheng:/app/data
|
||||
security_opt:
|
||||
- seccomp:unconfined
|
||||
command: sh entrypoint.sh worker # 启动celery的异步worker服务,用来处理一些耗时的任务
|
||||
restart: on-failure
|
||||
depends_on:
|
||||
mysql:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
office:
|
||||
condition: service_started
|
||||
|
||||
frontend:
|
||||
container_name: bisheng-frontend
|
||||
image: dataelement/bisheng-frontend:latest
|
||||
image: dataelement/bisheng-frontend:v1.3.1
|
||||
ports:
|
||||
- "3001:3001"
|
||||
environment:
|
||||
@@ -107,7 +151,7 @@ services:
|
||||
restart: on-failure
|
||||
|
||||
etcd:
|
||||
container_name: milvus-etcd
|
||||
container_name: bisheng-milvus-etcd
|
||||
image: quay.io/coreos/etcd:v3.5.5
|
||||
environment:
|
||||
ETCD_AUTO_COMPACTION_MODE: revision
|
||||
@@ -126,7 +170,7 @@ services:
|
||||
retries: 3
|
||||
|
||||
minio:
|
||||
container_name: milvus-minio
|
||||
container_name: bisheng-milvus-minio
|
||||
image: minio/minio:RELEASE.2023-03-20T20-16-18Z
|
||||
environment:
|
||||
MINIO_ACCESS_KEY: minioadmin
|
||||
@@ -146,8 +190,8 @@ services:
|
||||
retries: 3
|
||||
|
||||
milvus:
|
||||
container_name: milvus-standalone
|
||||
image: milvusdb/milvus:v2.3.3
|
||||
container_name: bisheng-milvus-standalone
|
||||
image: milvusdb/milvus:v2.5.10
|
||||
command: ["milvus", "run", "standalone"]
|
||||
security_opt:
|
||||
- seccomp:unconfined
|
||||
|
||||
@@ -18,14 +18,14 @@ server {
|
||||
|
||||
listen 3001;
|
||||
|
||||
location / {
|
||||
root /usr/share/nginx/html;
|
||||
location / {
|
||||
root /usr/share/nginx/html/platform;
|
||||
index index.html index.htm;
|
||||
try_files $uri $uri/ /index.html =404;
|
||||
add_header X-Frame-Options SAMEORIGIN;
|
||||
}
|
||||
|
||||
location /api {
|
||||
location /api {
|
||||
proxy_pass http://backend:7860;
|
||||
proxy_read_timeout 300s;
|
||||
proxy_set_header Host $host;
|
||||
@@ -39,7 +39,28 @@ server {
|
||||
add_header X-Frame-Options SAMEORIGIN;
|
||||
}
|
||||
|
||||
location /bisheng {
|
||||
location /workspace/ {
|
||||
alias /usr/share/nginx/html/client/;
|
||||
index index.html index.htm;
|
||||
try_files $uri $uri/ /workspace/index.html;
|
||||
}
|
||||
|
||||
location /workspace/api {
|
||||
rewrite ^/workspace(/.*)$ $1 break;
|
||||
proxy_pass http://backend:7860;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection $connection_upgrade;
|
||||
client_max_body_size 50m;
|
||||
add_header Access-Control-Allow-Origin $host;
|
||||
add_header X-Frame-Options SAMEORIGIN;
|
||||
}
|
||||
|
||||
location ~ ^/(workspace/bisheng|bisheng|tmp-dir)/ {
|
||||
rewrite ^/workspace(/.*)$ $1 break;
|
||||
proxy_pass http://minio:9000;
|
||||
}
|
||||
}
|
||||
+7
-20
@@ -1,27 +1,14 @@
|
||||
FROM python:3.10-slim
|
||||
FROM dataelement/bisheng-backend:base.v3
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Install Poetry
|
||||
RUN apt-get update && apt-get install gcc g++ curl build-essential postgresql-server-dev-all -y
|
||||
RUN apt-get update && apt-get install procps -y
|
||||
# Install font
|
||||
RUN apt install fonts-wqy-zenhei -y
|
||||
# opencv
|
||||
RUN apt-get install -y libglib2.0-0 libsm6 libxrender1 libxext6 libgl1
|
||||
RUN curl -sSL https://install.python-poetry.org | python3 - --version 1.8.2
|
||||
# # Add Poetry to PATH
|
||||
ENV PATH="${PATH}:/root/.local/bin"
|
||||
# # Copy the pyproject.toml and poetry.lock files
|
||||
# COPY poetry.lock pyproject.toml ./
|
||||
# Copy the rest of the application codes
|
||||
COPY ./ ./
|
||||
|
||||
RUN python -m pip install --upgrade pip && \
|
||||
pip install shapely==2.0.1
|
||||
|
||||
# Install dependencies
|
||||
RUN poetry config virtualenvs.create false
|
||||
RUN poetry install --no-interaction --no-ansi --without dev
|
||||
RUN poetry update --without dev
|
||||
|
||||
CMD ["uvicorn", "bisheng.main:app", "--workers", "2", "--host", "0.0.0.0", "--port", "7860"]
|
||||
# patch langchain-openai lib. remove this when langchain-openai support reasoning_content
|
||||
RUN patch -p1 < /app/bisheng/patches/langchain_openai.patch /usr/local/lib/python3.10/site-packages/langchain_openai/chat_models/base.py
|
||||
|
||||
|
||||
CMD ["sh entrypoint.sh"]
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
# 毕昇后端代码
|
||||
|
||||
* Dockerfile 使用 poetry 进行 Python 依赖管理
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
FROM python:3.10-slim
|
||||
|
||||
ARG PANDOC_ARCH=amd64
|
||||
ENV PANDOC_ARCH=$PANDOC_ARCH
|
||||
ENV PATH="${PATH}:/root/.local/bin"
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# 使用国内源 + 安装依赖(合并指令、清理缓存、禁用推荐包)
|
||||
RUN echo "\
|
||||
deb https://mirrors.aliyun.com/debian/ bookworm main non-free non-free-firmware contrib\n\
|
||||
deb https://mirrors.aliyun.com/debian-security/ bookworm-security main\n\
|
||||
deb https://mirrors.aliyun.com/debian/ bookworm-updates main non-free non-free-firmware contrib\n\
|
||||
deb https://mirrors.aliyun.com/debian/ bookworm-backports main non-free non-free-firmware contrib" \
|
||||
> /etc/apt/sources.list && \
|
||||
apt-get update && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
gcc g++ curl build-essential postgresql-server-dev-all libreoffice \
|
||||
wget procps vim fonts-wqy-zenhei \
|
||||
libglib2.0-0 libsm6 libxrender1 libxext6 libgl1 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
|
||||
# 安装 pandoc
|
||||
RUN mkdir -p /opt/pandoc && \
|
||||
cd /opt/pandoc && \
|
||||
wget https://github.com/jgm/pandoc/releases/download/3.6.4/pandoc-3.6.4-linux-${PANDOC_ARCH}.tar.gz && \
|
||||
tar xvf pandoc-3.6.4-linux-${PANDOC_ARCH}.tar.gz && \
|
||||
cp pandoc-3.6.4/bin/pandoc /usr/bin/ && \
|
||||
rm -rf /opt/pandoc
|
||||
|
||||
# 安装 Poetry
|
||||
RUN curl -sSL https://install.python-poetry.org | python3 - --version 1.8.2
|
||||
|
||||
# 拷贝项目依赖文件
|
||||
COPY ./pyproject.toml ./
|
||||
|
||||
# 安装 Python 依赖
|
||||
RUN python -m pip install --upgrade pip && \
|
||||
pip install shapely==2.0.1 && \
|
||||
poetry config virtualenvs.create false && \
|
||||
poetry install --no-interaction --no-ansi --without dev
|
||||
|
||||
# 安装 NLTK 数据
|
||||
RUN python -c "import nltk; nltk.download('punkt'); nltk.download('punkt_tab'); nltk.download('averaged_perceptron_tagger'); nltk.download('averaged_perceptron_tagger_eng')"
|
||||
|
||||
COPY . .
|
||||
|
||||
CMD ["sh", "entrypoint.sh"]
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
config.yaml
|
||||
@@ -5,11 +5,11 @@ from bisheng.interface.custom.custom_component import CustomComponent
|
||||
from bisheng.processing.process import load_flow_from_json # noqa: E402
|
||||
|
||||
try:
|
||||
__version__ = metadata.version(__package__)
|
||||
# 通过ci去自动修改
|
||||
__version__ = '1.3.1'
|
||||
except metadata.PackageNotFoundError:
|
||||
# Case where package metadata is not available.
|
||||
__version__ = ''
|
||||
del metadata # optional, avoids polluting the results of dir(__package__)
|
||||
|
||||
|
||||
__all__ = ['load_flow_from_json', 'cache_manager', 'CustomComponent']
|
||||
|
||||
@@ -1,8 +1,13 @@
|
||||
import json
|
||||
from typing import List
|
||||
|
||||
from bisheng.settings import settings
|
||||
from pydantic import BaseModel
|
||||
|
||||
from bisheng.settings import settings
|
||||
|
||||
# 配置JWT token的有效期
|
||||
ACCESS_TOKEN_EXPIRE_TIME = 86400
|
||||
|
||||
|
||||
class Settings(BaseModel):
|
||||
authjwt_secret_key: str = settings.jwt_secret
|
||||
@@ -10,3 +15,5 @@ class Settings(BaseModel):
|
||||
authjwt_token_location: List[str] = ['cookies', 'headers']
|
||||
# Disable CSRF Protection for this example. default is True
|
||||
authjwt_cookie_csrf_protect: bool = False
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
# 统一的错误码
|
||||
|
||||
## 错误码前三位代表具体功能模块,后两位表示模块内部具体的报错。例如10001。有新增请同步到前端做国际化
|
||||
|
||||
### 100 公共错误码
|
||||
|
||||
### 101 微调模块
|
||||
|
||||
### 103 个人组件模块
|
||||
|
||||
### 104 助手模块
|
||||
|
||||
### 105 技能模块
|
||||
|
||||
### 106 用户模块
|
||||
|
||||
### 107 标签模块
|
||||
|
||||
### 108 模型管理
|
||||
|
||||
### 109 知识库模块
|
||||
@@ -1,3 +1,5 @@
|
||||
from fastapi.exceptions import HTTPException
|
||||
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel
|
||||
|
||||
|
||||
@@ -10,7 +12,21 @@ class BaseErrorCode:
|
||||
def return_resp(cls, msg: str = None, data: any = None) -> UnifiedResponseModel:
|
||||
return UnifiedResponseModel(status_code=cls.Code, status_message=msg or cls.Msg, data=data)
|
||||
|
||||
@classmethod
|
||||
def http_exception(cls, msg: str = None) -> HTTPException:
|
||||
return HTTPException(status_code=cls.Code, detail=msg or cls.Msg)
|
||||
|
||||
|
||||
class UnAuthorizedError(BaseErrorCode):
|
||||
Code: int = 403
|
||||
Msg: str = '暂无操作权限'
|
||||
|
||||
|
||||
class NotFoundError(BaseErrorCode):
|
||||
Code: int = 404
|
||||
Msg: str = '资源不存在'
|
||||
|
||||
|
||||
class ServerError(BaseErrorCode):
|
||||
Code: int = 500
|
||||
Msg: str = '服务器错误'
|
||||
|
||||
@@ -65,3 +65,8 @@ class TrainFileNotExistError(BaseErrorCode):
|
||||
class GetGPUInfoError(BaseErrorCode):
|
||||
Code: int = 10125
|
||||
Msg: str = '获取GPU信息失败'
|
||||
|
||||
|
||||
class GetModelError(BaseErrorCode):
|
||||
Code: int = 10126
|
||||
Msg: str = '获取模型列表失败'
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
from bisheng.api.errcode.base import BaseErrorCode
|
||||
|
||||
|
||||
# RT服务相关的返回错误码,功能模块代码:105
|
||||
# 技能服务相关的返回错误码,功能模块代码:105
|
||||
class NotFoundVersionError(BaseErrorCode):
|
||||
Code: int = 10500
|
||||
Msg: str = '未找到技能版本信息'
|
||||
@@ -17,6 +17,11 @@ class VersionNameExistsError(BaseErrorCode):
|
||||
Msg: str = '版本名已存在'
|
||||
|
||||
|
||||
class FlowNameExistsError(BaseErrorCode):
|
||||
Code: int = 10503
|
||||
Msg: str = '技能名重复'
|
||||
|
||||
|
||||
class NotFoundFlowError(BaseErrorCode):
|
||||
Code: int = 10520
|
||||
Msg: str = '技能不存在'
|
||||
@@ -26,3 +31,47 @@ class FlowOnlineEditError(BaseErrorCode):
|
||||
Code: int = 10521
|
||||
Msg: str = '技能已上线,不可编辑'
|
||||
|
||||
|
||||
class WorkFlowOnlineEditError(BaseErrorCode):
|
||||
Code: int = 10525
|
||||
Msg: str = '工作流已上线,不可编辑'
|
||||
|
||||
|
||||
class WorkFlowInitError(BaseErrorCode):
|
||||
Code: int = 10526
|
||||
Msg: str = '工作流初始化失败'
|
||||
|
||||
|
||||
class WorkFlowWaitUserTimeoutError(BaseErrorCode):
|
||||
Code: int = 10527
|
||||
Msg: str = '工作流等待用户输入超时'
|
||||
|
||||
|
||||
class WorkFlowNodeRunMaxTimesError(BaseErrorCode):
|
||||
Code: int = 10528
|
||||
Msg: str = '节点执行超过最大次数'
|
||||
|
||||
|
||||
class WorkflowNameExistsError(BaseErrorCode):
|
||||
Code: int = 10529
|
||||
Msg: str = '工作流名称重复'
|
||||
|
||||
|
||||
class FlowTemplateNameError(BaseErrorCode):
|
||||
Code: int = 10530
|
||||
Msg: str = '模板名称已存在'
|
||||
|
||||
|
||||
class WorkFlowNodeUpdateError(BaseErrorCode):
|
||||
Code: int = 10531
|
||||
Msg: str = '<节点名称>功能已升级,需删除后重新拖入。'
|
||||
|
||||
|
||||
class WorkFlowVersionUpdateError(BaseErrorCode):
|
||||
Code: int = 10532
|
||||
Msg: str = '工作流版本已升级,请联系创建者重新编排'
|
||||
|
||||
|
||||
class WorkFlowTaskBusyError(BaseErrorCode):
|
||||
Code: int = 10540
|
||||
Msg: str = '服务器线程数已满,请稍候再试'
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
from bisheng.api.errcode.base import BaseErrorCode
|
||||
|
||||
|
||||
# 知识库模块相关的返回错误码,功能模块代码:109
|
||||
class KnowledgeExistError(BaseErrorCode):
|
||||
Code: int = 10900
|
||||
Msg: str = '知识库名称重复'
|
||||
|
||||
|
||||
class KnowledgeNoEmbeddingError(BaseErrorCode):
|
||||
Code: int = 10901
|
||||
Msg: str = '知识库必须选择一个embedding模型'
|
||||
|
||||
|
||||
class KnowledgeChunkError(BaseErrorCode):
|
||||
Code: int = 10910
|
||||
Msg: str = '当前知识库版本不支持修改分段,请创建新知识库后进行分段修改'
|
||||
|
||||
|
||||
class KnowledgeSimilarError(BaseErrorCode):
|
||||
Code: int = 10920
|
||||
Msg: str = '未配置QA知识库相似问模型'
|
||||
|
||||
|
||||
class KnowledgeQAError(BaseErrorCode):
|
||||
Code: int = 10930
|
||||
Msg: str = '该问题已存在'
|
||||
|
||||
|
||||
class KnowledgeCPError(BaseErrorCode):
|
||||
Code: int = 10940
|
||||
Msg: str = '当前有文件正在解析,不可复制'
|
||||
@@ -0,0 +1,22 @@
|
||||
from bisheng.api.errcode.base import BaseErrorCode
|
||||
|
||||
|
||||
# 模型管理模块相关的返回错误码,功能模块代码:108
|
||||
class ServerExistError(BaseErrorCode):
|
||||
Code: int = 10800
|
||||
Msg: str = '服务提供方名称重复,请修改'
|
||||
|
||||
|
||||
class ModelNameRepeatError(BaseErrorCode):
|
||||
Code: int = 10801
|
||||
Msg: str = '模型不可重复'
|
||||
|
||||
|
||||
class ServerAddAllError(BaseErrorCode):
|
||||
Code: int = 10802
|
||||
Msg: str = '添加服务提供方失败,模型全部初始化失败'
|
||||
|
||||
|
||||
class ServerAddError(BaseErrorCode):
|
||||
Code: int = 10803
|
||||
Msg: str = '添加服务提供方失败,部分模型初始化失败'
|
||||
@@ -0,0 +1,12 @@
|
||||
from bisheng.api.errcode.base import BaseErrorCode
|
||||
|
||||
|
||||
# 标签模块相关的返回错误码,功能模块代码:107
|
||||
class TagExistError(BaseErrorCode):
|
||||
Code: int = 10700
|
||||
Msg: str = '标签已存在'
|
||||
|
||||
|
||||
class TagNotExistError(BaseErrorCode):
|
||||
Code: int = 10701
|
||||
Msg: str = '未找到对应的标签'
|
||||
@@ -0,0 +1,42 @@
|
||||
from bisheng.api.errcode.base import BaseErrorCode
|
||||
|
||||
|
||||
# 用户模块相关的返回错误码,功能模块代码:106
|
||||
class UserValidateError(BaseErrorCode):
|
||||
Code: int = 10600
|
||||
Msg: str = '账号或密码错误'
|
||||
|
||||
|
||||
class UserPasswordExpireError(BaseErrorCode):
|
||||
Code: int = 10601
|
||||
Msg: str = '您的密码已过期,请及时修改'
|
||||
|
||||
|
||||
class UserNotPasswordError(BaseErrorCode):
|
||||
Code: int = 10602
|
||||
Msg: str = '用户尚未设置密码,请先联系管理员重置密码'
|
||||
|
||||
|
||||
class UserPasswordError(BaseErrorCode):
|
||||
Code: int = 10603
|
||||
Msg: str = '当前密码错误'
|
||||
|
||||
|
||||
class UserLoginOfflineError(BaseErrorCode):
|
||||
Code: int = 10604
|
||||
Msg: str = '您的账户已在另一设备上登录,此设备上的会话已被注销。\n如果这不是您本人的操作,请尽快修改您的账户密码。'
|
||||
|
||||
|
||||
class UserNameAlreadyExistError(BaseErrorCode):
|
||||
Code: int = 10605
|
||||
Msg: str = '用户名已存在'
|
||||
|
||||
|
||||
class UserNeedGroupAndRoleError(BaseErrorCode):
|
||||
Code: int = 10606
|
||||
Msg: str = '用户组和角色不能为空'
|
||||
|
||||
|
||||
class UserGroupNotDeleteError(BaseErrorCode):
|
||||
Code: int = 10610
|
||||
Msg: str = '用户组内还有用户,不能删除'
|
||||
@@ -1,11 +1,16 @@
|
||||
# Router for base api
|
||||
from bisheng.api.v1 import (assistant_router, chat_router, component_router, endpoints_router,
|
||||
finetune_router, flows_router, knowledge_router, qa_router,
|
||||
report_router, server_router, skillcenter_router, user_router,
|
||||
validate_router, variable_router)
|
||||
from bisheng.api.v2 import chat_router_rpc, knowledge_router_rpc, rpc_router_rpc
|
||||
from fastapi import APIRouter
|
||||
|
||||
from bisheng.api.v1 import (assistant_router, audit_router, chat_router, component_router,
|
||||
endpoints_router, evaluation_router, finetune_router, flows_router,
|
||||
group_router, knowledge_router, llm_router, mark_router, qa_router,
|
||||
report_router, server_router, skillcenter_router, tag_router,
|
||||
user_router, validate_router, variable_router, workflow_router,
|
||||
workstation_router, linsight_router, tool_router, invite_code_router)
|
||||
from bisheng.api.v2 import (assistant_router_rpc, chat_router_rpc, flow_router,
|
||||
knowledge_router_rpc, rpc_router_rpc, workflow_router_rpc,
|
||||
workstation_router_rpc)
|
||||
|
||||
router = APIRouter(prefix='/api/v1', )
|
||||
router.include_router(chat_router)
|
||||
router.include_router(endpoints_router)
|
||||
@@ -21,8 +26,22 @@ router.include_router(report_router)
|
||||
router.include_router(finetune_router)
|
||||
router.include_router(component_router)
|
||||
router.include_router(assistant_router)
|
||||
|
||||
router.include_router(group_router)
|
||||
router.include_router(audit_router)
|
||||
router.include_router(evaluation_router)
|
||||
router.include_router(tag_router)
|
||||
router.include_router(llm_router)
|
||||
router.include_router(workflow_router)
|
||||
router.include_router(mark_router)
|
||||
router.include_router(workstation_router)
|
||||
router.include_router(linsight_router)
|
||||
router.include_router(tool_router)
|
||||
router.include_router(invite_code_router)
|
||||
router_rpc = APIRouter(prefix='/api/v2', )
|
||||
router_rpc.include_router(knowledge_router_rpc)
|
||||
router_rpc.include_router(chat_router_rpc)
|
||||
router_rpc.include_router(rpc_router_rpc)
|
||||
router_rpc.include_router(flow_router)
|
||||
router_rpc.include_router(assistant_router_rpc)
|
||||
router_rpc.include_router(workflow_router_rpc)
|
||||
router_rpc.include_router(workstation_router_rpc)
|
||||
|
||||
@@ -1,29 +1,40 @@
|
||||
import json
|
||||
from datetime import datetime
|
||||
from typing import Any, List, Optional
|
||||
from uuid import UUID
|
||||
|
||||
from fastapi import Request
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.errcode.assistant import (AssistantInitError, AssistantNameRepeatError,
|
||||
AssistantNotEditError, AssistantNotExistsError, ToolTypeRepeatError,
|
||||
ToolTypeEmptyError, ToolTypeNotExistsError, ToolTypeIsPresetError)
|
||||
from bisheng.api.errcode.base import UnAuthorizedError
|
||||
ToolTypeIsPresetError)
|
||||
from bisheng.api.errcode.base import UnAuthorizedError, NotFoundError
|
||||
from bisheng.api.services.assistant_agent import AssistantAgent
|
||||
from bisheng.api.services.assistant_base import AssistantUtils
|
||||
from bisheng.api.services.audit_log import AuditLogService
|
||||
from bisheng.api.services.base import BaseService
|
||||
from bisheng.api.services.llm import LLMService
|
||||
from bisheng.api.services.tool import ToolServices
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.utils import get_request_ip
|
||||
from bisheng.api.v1.schemas import (AssistantInfo, AssistantSimpleInfo, AssistantUpdateReq,
|
||||
StreamData, UnifiedResponseModel, resp_200, resp_500)
|
||||
from bisheng.cache import InMemoryCache
|
||||
from bisheng.database.constants import ToolPresetType
|
||||
from bisheng.database.models.assistant import (Assistant, AssistantDao, AssistantLinkDao,
|
||||
AssistantStatus)
|
||||
from bisheng.database.models.flow import Flow, FlowDao
|
||||
from bisheng.database.models.gpts_tools import GptsToolsDao, GptsToolsRead, GptsToolsTypeRead, GptsTools
|
||||
from bisheng.database.models.gpts_tools import GptsToolsDao, GptsToolsTypeRead, GptsTools
|
||||
from bisheng.database.models.group_resource import GroupResourceDao, GroupResource, ResourceTypeEnum
|
||||
from bisheng.database.models.knowledge import KnowledgeDao
|
||||
from bisheng.database.models.role_access import AccessType, RoleAccessDao
|
||||
from bisheng.database.models.tag import TagDao
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.database.models.user_group import UserGroupDao
|
||||
from bisheng.database.models.user_role import UserRoleDao
|
||||
from loguru import logger
|
||||
|
||||
|
||||
class AssistantService(AssistantUtils):
|
||||
class AssistantService(BaseService, AssistantUtils):
|
||||
UserCache: InMemoryCache = InMemoryCache()
|
||||
|
||||
@classmethod
|
||||
@@ -31,14 +42,25 @@ class AssistantService(AssistantUtils):
|
||||
user: UserPayload,
|
||||
name: str = None,
|
||||
status: int | None = None,
|
||||
tag_id: int | None = None,
|
||||
page: int = 1,
|
||||
limit: int = 20) -> UnifiedResponseModel[List[AssistantSimpleInfo]]:
|
||||
"""
|
||||
获取助手列表
|
||||
"""
|
||||
assistant_ids = []
|
||||
if tag_id:
|
||||
ret = TagDao.get_resources_by_tags([tag_id], ResourceTypeEnum.ASSISTANT)
|
||||
assistant_ids = [one.resource_id for one in ret]
|
||||
if not assistant_ids:
|
||||
return resp_200(data={
|
||||
'data': [],
|
||||
'total': 0
|
||||
})
|
||||
|
||||
data = []
|
||||
if user.is_admin():
|
||||
res, total = AssistantDao.get_all_assistants(name, page, limit)
|
||||
res, total = AssistantDao.get_all_assistants(name, page, limit, assistant_ids, status)
|
||||
else:
|
||||
# 权限管理可见的助手信息
|
||||
assistant_ids_extra = []
|
||||
@@ -47,24 +69,52 @@ class AssistantService(AssistantUtils):
|
||||
role_ids = [role.role_id for role in user_role]
|
||||
role_access = RoleAccessDao.get_role_access(role_ids, AccessType.ASSISTANT_READ)
|
||||
if role_access:
|
||||
assistant_ids_extra = [UUID(access.third_id).hex for access in role_access]
|
||||
res, total = AssistantDao.get_assistants(user.user_id, name, assistant_ids_extra, status, page, limit)
|
||||
assistant_ids_extra = [access.third_id for access in role_access]
|
||||
res, total = AssistantDao.get_assistants(user.user_id, name, assistant_ids_extra, status, page, limit,
|
||||
assistant_ids)
|
||||
|
||||
assistant_ids = [one.id for one in res]
|
||||
# 查询助手所属的分组
|
||||
assistant_groups = GroupResourceDao.get_resources_group(ResourceTypeEnum.ASSISTANT, assistant_ids)
|
||||
assistant_group_dict = {}
|
||||
for one in assistant_groups:
|
||||
if one.third_id not in assistant_group_dict:
|
||||
assistant_group_dict[one.third_id] = []
|
||||
assistant_group_dict[one.third_id].append(one.group_id)
|
||||
|
||||
# 获取助手关联的tag
|
||||
flow_tags = TagDao.get_tags_by_resource(ResourceTypeEnum.ASSISTANT, assistant_ids)
|
||||
|
||||
for one in res:
|
||||
simple_dict = one.model_dump(include={
|
||||
'id', 'name', 'desc', 'logo', 'status', 'user_id', 'create_time', 'update_time'
|
||||
})
|
||||
one.logo = cls.get_logo_share_link(one.logo)
|
||||
simple_assistant = cls.return_simple_assistant_info(one)
|
||||
if one.user_id == user.user_id or user.is_admin():
|
||||
simple_dict['write'] = True
|
||||
simple_dict['user_name'] = cls.get_user_name(one.user_id)
|
||||
data.append(AssistantSimpleInfo(**simple_dict))
|
||||
simple_assistant.write = True
|
||||
simple_assistant.group_ids = assistant_group_dict.get(one.id, [])
|
||||
simple_assistant.tags = flow_tags.get(one.id, [])
|
||||
data.append(simple_assistant)
|
||||
return resp_200(data={'data': data, 'total': total})
|
||||
|
||||
@classmethod
|
||||
def get_assistant_info(cls, assistant_id: UUID, user_id: str):
|
||||
def return_simple_assistant_info(cls, one: Assistant) -> AssistantSimpleInfo:
|
||||
"""
|
||||
将数据库的 助手model简化 处理后成返回前端的格式
|
||||
"""
|
||||
simple_dict = one.model_dump(include={
|
||||
'id', 'name', 'desc', 'logo', 'status', 'user_id', 'create_time', 'update_time'
|
||||
})
|
||||
simple_dict['user_name'] = cls.get_user_name(one.user_id)
|
||||
return AssistantSimpleInfo(**simple_dict)
|
||||
|
||||
@classmethod
|
||||
def get_assistant_info(cls, assistant_id: str, login_user: UserPayload):
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
if not assistant:
|
||||
if not assistant or assistant.is_delete:
|
||||
return AssistantNotExistsError.return_resp()
|
||||
# 检查是否有权限获取信息
|
||||
if not login_user.access_check(assistant.user_id, assistant.id, AccessType.ASSISTANT_READ):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
tool_list = []
|
||||
flow_list = []
|
||||
knowledge_list = []
|
||||
@@ -81,6 +131,7 @@ class AssistantService(AssistantUtils):
|
||||
logger.error(f'not expect link info: {one.dict()}')
|
||||
tool_list, flow_list, knowledge_list = cls.get_link_info(tool_list, flow_list,
|
||||
knowledge_list)
|
||||
assistant.logo = cls.get_logo_share_link(assistant.logo)
|
||||
return resp_200(data=AssistantInfo(**assistant.dict(),
|
||||
tool_list=tool_list,
|
||||
flow_list=flow_list,
|
||||
@@ -88,51 +139,92 @@ class AssistantService(AssistantUtils):
|
||||
|
||||
# 创建助手
|
||||
@classmethod
|
||||
async def create_assistant(cls, assistant: Assistant) -> UnifiedResponseModel[AssistantInfo]:
|
||||
async def create_assistant(cls, request: Request, login_user: UserPayload, assistant: Assistant) \
|
||||
-> UnifiedResponseModel[AssistantInfo]:
|
||||
|
||||
# 检查下是否有重名
|
||||
if cls.judge_name_repeat(assistant.name, assistant.user_id):
|
||||
return AssistantNameRepeatError.return_resp()
|
||||
|
||||
# 保存数据到数据库, 补充用默认的模型
|
||||
llm_conf = cls.get_llm_conf(assistant.model_name)
|
||||
assistant.model_name = llm_conf['model_name']
|
||||
assistant.temperature = llm_conf['temperature']
|
||||
|
||||
logger.info(f"assistant original prompt id: {assistant.id}, desc: {assistant.prompt}")
|
||||
|
||||
# 自动补充默认的模型配置
|
||||
assistant_llm = LLMService.get_assistant_llm()
|
||||
if assistant_llm.llm_list:
|
||||
for one in assistant_llm.llm_list:
|
||||
if one.default:
|
||||
assistant.model_name = one.model_id
|
||||
break
|
||||
|
||||
# 自动生成描述
|
||||
assistant, _, _ = await cls.get_auto_info(assistant)
|
||||
assistant = AssistantDao.create_assistant(assistant)
|
||||
|
||||
cls.create_assistant_hook(request, assistant, login_user)
|
||||
return resp_200(data=AssistantInfo(**assistant.dict(),
|
||||
tool_list=[],
|
||||
flow_list=[],
|
||||
knowledge_list=[]))
|
||||
|
||||
@classmethod
|
||||
def create_assistant_hook(cls, request: Request, assistant: Assistant, user_payload: UserPayload) -> bool:
|
||||
"""
|
||||
创建助手成功后的hook,执行一些其他业务逻辑
|
||||
"""
|
||||
# 查询下用户所在的用户组
|
||||
user_group = UserGroupDao.get_user_group(user_payload.user_id)
|
||||
if user_group:
|
||||
# 批量将助手资源插入到关联表里
|
||||
batch_resource = []
|
||||
for one in user_group:
|
||||
batch_resource.append(GroupResource(
|
||||
group_id=one.group_id,
|
||||
third_id=assistant.id,
|
||||
type=ResourceTypeEnum.ASSISTANT.value))
|
||||
GroupResourceDao.insert_group_batch(batch_resource)
|
||||
|
||||
# 写入审计日志
|
||||
AuditLogService.create_build_assistant(user_payload, get_request_ip(request), assistant.id)
|
||||
|
||||
# 写入logo缓存
|
||||
cls.get_logo_share_link(assistant.logo)
|
||||
return True
|
||||
|
||||
# 删除助手
|
||||
@classmethod
|
||||
def delete_assistant(cls, assistant_id: UUID, user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
def delete_assistant(cls, request: Request, login_user: UserPayload, assistant_id: str) -> UnifiedResponseModel:
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
if not assistant:
|
||||
return AssistantNotExistsError.return_resp()
|
||||
|
||||
# 判断授权
|
||||
if not user_payload.access_check(assistant.user_id, assistant.id.hex, AccessType.ASSISTANT_WRITE):
|
||||
if not login_user.access_check(assistant.user_id, assistant.id, AccessType.ASSISTANT_WRITE):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
AssistantDao.delete_assistant(assistant)
|
||||
cls.delete_assistant_hook(request, login_user, assistant)
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
async def auto_update_stream(cls, assistant_id: UUID, prompt: str):
|
||||
def delete_assistant_hook(cls, request: Request, login_user: UserPayload, assistant: Assistant) -> bool:
|
||||
""" 清理关联的助手资源 """
|
||||
logger.info(f"delete_assistant_hook id: {assistant.id}, user: {login_user.user_id}")
|
||||
# 写入审计日志
|
||||
AuditLogService.delete_build_assistant(login_user, get_request_ip(request), assistant.id)
|
||||
|
||||
# 清理和用户组的关联
|
||||
GroupResourceDao.delete_group_resource_by_third_id(assistant.id, ResourceTypeEnum.ASSISTANT)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
async def auto_update_stream(cls, assistant_id: str, prompt: str):
|
||||
""" 重新生成助手的提示词和工具选择, 只调用模型能力不修改数据库数据 """
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
assistant.prompt = prompt
|
||||
|
||||
# 初始化llm
|
||||
auto_agent = AssistantAgent(assistant, '')
|
||||
await auto_agent.init_llm()
|
||||
await auto_agent.init_auto_update_llm()
|
||||
|
||||
# 流式生成提示词
|
||||
final_prompt = ''
|
||||
@@ -162,14 +254,14 @@ class AssistantService(AssistantUtils):
|
||||
yield str(StreamData(event='message', data={'type': 'flow_list', 'message': flow_info}))
|
||||
|
||||
@classmethod
|
||||
async def update_assistant(cls, req: AssistantUpdateReq, user_payload: UserPayload) \
|
||||
async def update_assistant(cls, request: Request, login_user: UserPayload, req: AssistantUpdateReq) \
|
||||
-> UnifiedResponseModel[AssistantInfo]:
|
||||
""" 更新助手信息 """
|
||||
assistant = AssistantDao.get_one_assistant(req.id)
|
||||
if not assistant:
|
||||
return AssistantNotExistsError.return_resp()
|
||||
|
||||
check_result = cls.check_update_permission(assistant, user_payload)
|
||||
check_result = cls.check_update_permission(assistant, login_user)
|
||||
if check_result is not None:
|
||||
return check_result
|
||||
|
||||
@@ -187,6 +279,7 @@ class AssistantService(AssistantUtils):
|
||||
assistant.model_name = req.model_name
|
||||
assistant.temperature = req.temperature
|
||||
assistant.update_time = datetime.now()
|
||||
assistant.max_token = req.max_token
|
||||
AssistantDao.update_assistant(assistant)
|
||||
|
||||
# 更新助手关联信息
|
||||
@@ -196,25 +289,38 @@ class AssistantService(AssistantUtils):
|
||||
AssistantLinkDao.update_assistant_flow(assistant.id, flow_list=req.flow_list)
|
||||
if req.knowledge_list is not None:
|
||||
# 使用配置的flow 进行技能补充
|
||||
flow_id_default = AssistantUtils.get_default_retrieval()
|
||||
AssistantLinkDao.update_assistant_knowledge(assistant.id,
|
||||
knowledge_list=req.knowledge_list,
|
||||
flow_id=flow_id_default)
|
||||
flow_id='')
|
||||
tool_list, flow_list, knowledge_list = cls.get_link_info(req.tool_list, req.flow_list,
|
||||
req.knowledge_list)
|
||||
cls.update_assistant_hook(request, login_user, assistant)
|
||||
return resp_200(data=AssistantInfo(**assistant.dict(),
|
||||
tool_list=tool_list,
|
||||
flow_list=flow_list,
|
||||
knowledge_list=knowledge_list))
|
||||
|
||||
@classmethod
|
||||
async def update_status(cls, assistant_id: UUID, status: int, user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
def update_assistant_hook(cls, request: Request, login_user: UserPayload, assistant: Assistant) -> bool:
|
||||
""" 更新助手的钩子 """
|
||||
logger.info(f"delete_assistant_hook id: {assistant.id}, user: {login_user.user_id}")
|
||||
|
||||
# 写入审计日志
|
||||
AuditLogService.update_build_assistant(login_user, get_request_ip(request), assistant.id)
|
||||
|
||||
# 写入缓存
|
||||
cls.get_logo_share_link(assistant.logo)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
async def update_status(cls, request: Request, login_user: UserPayload, assistant_id: str,
|
||||
status: int) -> UnifiedResponseModel:
|
||||
""" 更新助手的状态 """
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
if not assistant:
|
||||
return AssistantNotExistsError.return_resp()
|
||||
# 判断权限
|
||||
if not user_payload.access_check(assistant.user_id, assistant.id.hex, AccessType.ASSISTANT_WRITE):
|
||||
if not login_user.access_check(assistant.user_id, assistant.id, AccessType.ASSISTANT_WRITE):
|
||||
return UnAuthorizedError.return_resp()
|
||||
# 状态相等不做改动
|
||||
if assistant.status == status:
|
||||
@@ -230,10 +336,11 @@ class AssistantService(AssistantUtils):
|
||||
return AssistantInitError.return_resp('助手编译报错:' + str(e))
|
||||
assistant.status = status
|
||||
AssistantDao.update_assistant(assistant)
|
||||
cls.update_assistant_hook(request, login_user, assistant)
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
def update_prompt(cls, assistant_id: UUID, prompt: str, user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
def update_prompt(cls, assistant_id: str, prompt: str, user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
""" 更新助手的提示词 """
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
if not assistant:
|
||||
@@ -248,7 +355,7 @@ class AssistantService(AssistantUtils):
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
def update_flow_list(cls, assistant_id: UUID, flow_list: List[str],
|
||||
def update_flow_list(cls, assistant_id: str, flow_list: List[str],
|
||||
user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
""" 更新助手的技能列表 """
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
@@ -263,10 +370,28 @@ class AssistantService(AssistantUtils):
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
def get_gpts_tools(cls, user_id: Any, is_preset: Optional[bool] = None) -> List[GptsToolsTypeRead]:
|
||||
def get_gpts_tools(cls, user: UserPayload, is_preset: Optional[int] = None) -> List[GptsToolsTypeRead]:
|
||||
""" 获取用户可见的工具列表 """
|
||||
# 获取用户可见的工具类别
|
||||
all_tool_type = GptsToolsDao.get_tool_type(user_id, is_preset)
|
||||
tool_type_ids_extra = []
|
||||
if is_preset != ToolPresetType.PRESET.value:
|
||||
# 获取自定义工具列表时,需要包含用户可用的工具列表
|
||||
user_role = UserRoleDao.get_user_roles(user.user_id)
|
||||
if user_role:
|
||||
role_ids = [role.role_id for role in user_role]
|
||||
role_access = RoleAccessDao.get_role_access(role_ids, AccessType.GPTS_TOOL_READ)
|
||||
if role_access:
|
||||
tool_type_ids_extra = [int(access.third_id) for access in role_access]
|
||||
# 获取用户可见的所有工具列表
|
||||
if is_preset is None:
|
||||
all_tool_type = GptsToolsDao.get_user_tool_type(user.user_id, tool_type_ids_extra)
|
||||
elif is_preset == ToolPresetType.PRESET.value:
|
||||
# 获取预置工具列表
|
||||
all_tool_type = GptsToolsDao.get_preset_tool_type()
|
||||
else:
|
||||
# 获取用户可见的自定义工具列表
|
||||
all_tool_type = GptsToolsDao.get_user_tool_type(user.user_id, tool_type_ids_extra, False,
|
||||
ToolPresetType(is_preset))
|
||||
tool_type_id = [one.id for one in all_tool_type]
|
||||
res = []
|
||||
tool_type_children = {}
|
||||
@@ -281,104 +406,75 @@ class AssistantService(AssistantUtils):
|
||||
tool_type_children[one.type].append(one)
|
||||
|
||||
for one in res:
|
||||
one['write'] = one['id'] not in tool_type_ids_extra or one['user_id'] == user.user_id
|
||||
if not user.is_admin():
|
||||
one['extra'] = ''
|
||||
one["children"] = tool_type_children.get(one["id"], [])
|
||||
if one['extra']:
|
||||
extra = json.loads(one['extra'])
|
||||
one["parameter_name"] = extra.get("parameter_name")
|
||||
one["api_location"] = extra.get("api_location")
|
||||
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
def add_gpts_tools(cls, user: UserPayload, req: GptsToolsTypeRead) -> UnifiedResponseModel:
|
||||
def update_tool_config(cls, login_user: UserPayload, tool_type_id: int, extra: dict) -> GptsToolsTypeRead:
|
||||
# 获取工具类别
|
||||
tool_type = GptsToolsDao.get_one_tool_type(tool_type_id)
|
||||
if not tool_type:
|
||||
raise NotFoundError.http_exception()
|
||||
|
||||
# 更新工具类别下所有工具的配置
|
||||
tool_type.extra = json.dumps(extra, ensure_ascii=False)
|
||||
GptsToolsDao.update_tools_extra(tool_type_id, tool_type.extra)
|
||||
return tool_type
|
||||
|
||||
@classmethod
|
||||
async def add_gpts_tools(cls, user: UserPayload, req: GptsToolsTypeRead) -> UnifiedResponseModel:
|
||||
""" 添加自定义工具 """
|
||||
# 尝试解析下openapi schema看下是否可以正常解析, 不能的话保存不允许保存
|
||||
tool_service = ToolServices()
|
||||
if req.is_preset == ToolPresetType.API.value:
|
||||
await tool_service.parse_openapi_schema('', req.openapi_schema)
|
||||
elif req.is_preset == ToolPresetType.MCP.value:
|
||||
await tool_service.parse_mcp_schema(req.openapi_schema)
|
||||
|
||||
req.id = None
|
||||
if req.name.__len__() > 30 or req.name.__len__() == 0:
|
||||
return resp_500(message="名字不符合规范:至少1个字符,不能超过30个字符")
|
||||
if req.name.__len__() > 1000 or req.name.__len__() == 0:
|
||||
return resp_500(message="名字不符合规范:至少1个字符,不能超过1000个字符")
|
||||
# 判断类别是否已存在
|
||||
tool_type = GptsToolsDao.get_one_tool_type_by_name(user.user_id, req.name)
|
||||
if tool_type:
|
||||
return ToolTypeRepeatError.return_resp()
|
||||
if len(req.children) == 0:
|
||||
return ToolTypeEmptyError.return_resp()
|
||||
req.user_id = user.user_id
|
||||
|
||||
for one in req.children:
|
||||
one.id = None
|
||||
one.user_id = user.user_id
|
||||
one.is_delete = 0
|
||||
one.is_preset = False
|
||||
one.is_preset = req.is_preset
|
||||
|
||||
# 添加工具类别和对应的 工具列表
|
||||
res = GptsToolsDao.insert_tool_type(req)
|
||||
|
||||
cls.add_gpts_tools_hook(user, res)
|
||||
return resp_200(data=res)
|
||||
|
||||
@classmethod
|
||||
def update_gpts_tools(cls, user: UserPayload, req: GptsToolsTypeRead) -> UnifiedResponseModel:
|
||||
"""
|
||||
更新工具类别,包括更新工具类别的名称和删除、新增工具类别的API
|
||||
"""
|
||||
exist_tool_type = GptsToolsDao.get_one_tool_type(req.id)
|
||||
if not exist_tool_type:
|
||||
return ToolTypeNotExistsError.return_resp()
|
||||
if len(req.children) == 0:
|
||||
return ToolTypeEmptyError.return_resp()
|
||||
if req.name.__len__() > 30 or req.name.__len__() == 0:
|
||||
return resp_500(message="名字不符合规范:最少一个字符,不能超过30个字符")
|
||||
|
||||
# 判断工具类别名称是否重复
|
||||
tool_type = GptsToolsDao.get_one_tool_type_by_name(user.user_id, req.name)
|
||||
if tool_type and tool_type.id != exist_tool_type.id:
|
||||
return ToolTypeRepeatError.return_resp()
|
||||
|
||||
exist_tool_type.name = req.name
|
||||
exist_tool_type.logo = req.logo
|
||||
exist_tool_type.description = req.description
|
||||
exist_tool_type.server_host = req.server_host
|
||||
exist_tool_type.auth_method = req.auth_method
|
||||
exist_tool_type.api_key = req.api_key
|
||||
exist_tool_type.auth_type = req.auth_type
|
||||
exist_tool_type.openapi_schema = req.openapi_schema
|
||||
|
||||
children_map = {}
|
||||
for one in req.children:
|
||||
save_key = GptsToolsDao.get_tool_key(exist_tool_type.id, one.tool_key)
|
||||
save_key_prefix = save_key.split("_")[0]
|
||||
if one.tool_key.startswith(save_key_prefix):
|
||||
# 说明api和数据库的一致,没有通过openapiSchema重新解析
|
||||
children_map[one.tool_key] = one
|
||||
else:
|
||||
children_map[save_key] = one
|
||||
|
||||
# 获取此类别下旧的API列表
|
||||
old_tool_list = GptsToolsDao.get_list_by_type([exist_tool_type.id])
|
||||
# 需要被删除的工具列表
|
||||
delete_tool_id_list = []
|
||||
# 需要被更新的工具列表
|
||||
update_tool_list = []
|
||||
for one in old_tool_list:
|
||||
# 说明此工具 需要删除
|
||||
if children_map.get(one.tool_key) is None:
|
||||
delete_tool_id_list.append(one.id)
|
||||
else:
|
||||
# 说明此工具需要更新
|
||||
new_tool_info = children_map.pop(one.tool_key)
|
||||
one.name = new_tool_info.name
|
||||
one.desc = new_tool_info.desc
|
||||
one.extra = new_tool_info.extra
|
||||
one.api_params = new_tool_info.api_params
|
||||
update_tool_list.append(one)
|
||||
|
||||
add_children = []
|
||||
for one in children_map.values():
|
||||
one.id = None
|
||||
one.user_id = user.user_id
|
||||
one.is_preset = False
|
||||
one.is_delete = 0
|
||||
add_children.append(one)
|
||||
|
||||
GptsToolsDao.update_tool_type(exist_tool_type, delete_tool_id_list,
|
||||
add_children, update_tool_list)
|
||||
|
||||
children = GptsToolsDao.get_list_by_type([exist_tool_type.id])
|
||||
res = GptsToolsTypeRead(**exist_tool_type.model_dump(), children=children)
|
||||
return resp_200(data=res)
|
||||
def add_gpts_tools_hook(cls, user: UserPayload, gpts_tool_type: GptsToolsTypeRead) -> bool:
|
||||
""" 添加自定义工具后的hook函数 """
|
||||
# 查询下用户所在的用户组
|
||||
user_group = UserGroupDao.get_user_group(user.user_id)
|
||||
if user_group:
|
||||
# 批量将自定义工具插入到关联表里
|
||||
batch_resource = []
|
||||
for one in user_group:
|
||||
batch_resource.append(GroupResource(
|
||||
group_id=one.group_id,
|
||||
third_id=gpts_tool_type.id,
|
||||
type=ResourceTypeEnum.GPTS_TOOL.value))
|
||||
GroupResourceDao.insert_group_batch(batch_resource)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def delete_gpts_tools(cls, user: UserPayload, tool_type_id: int) -> UnifiedResponseModel:
|
||||
@@ -386,21 +482,25 @@ class AssistantService(AssistantUtils):
|
||||
exist_tool_type = GptsToolsDao.get_one_tool_type(tool_type_id)
|
||||
if not exist_tool_type:
|
||||
return resp_200()
|
||||
if exist_tool_type.is_preset:
|
||||
if exist_tool_type.is_preset == ToolPresetType.PRESET.value:
|
||||
return ToolTypeIsPresetError.return_resp()
|
||||
# 判断是否有更新权限
|
||||
if not user.access_check(exist_tool_type.user_id, str(exist_tool_type.id), AccessType.GPTS_TOOL_WRITE):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
GptsToolsDao.delete_tool_type(tool_type_id)
|
||||
cls.delete_gpts_tool_hook(user, exist_tool_type)
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
def get_models(cls) -> UnifiedResponseModel:
|
||||
llm_list = cls.get_gpts_conf('llms')
|
||||
res = []
|
||||
for one in llm_list:
|
||||
res.append({'id': one['model_name'], 'model_name': one['model_name']})
|
||||
return resp_200(data=res)
|
||||
def delete_gpts_tool_hook(cls, user: UserPayload, gpts_tool_type) -> bool:
|
||||
""" 删除自定义工具后的hook函数 """
|
||||
logger.info(f"delete_gpts_tool_hook id: {gpts_tool_type.id}, user: {user.user_id}")
|
||||
GroupResourceDao.delete_group_resource_by_third_id(gpts_tool_type.id, ResourceTypeEnum.GPTS_TOOL)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def update_tool_list(cls, assistant_id: UUID, tool_list: List[int],
|
||||
def update_tool_list(cls, assistant_id: str, tool_list: List[int],
|
||||
user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
""" 更新助手的工具列表 """
|
||||
assistant = AssistantDao.get_one_assistant(assistant_id)
|
||||
@@ -417,8 +517,8 @@ class AssistantService(AssistantUtils):
|
||||
@classmethod
|
||||
def check_update_permission(cls, assistant: Assistant, user_payload: UserPayload) -> Any:
|
||||
# 判断权限
|
||||
if not user_payload.access_check(assistant.user_id, assistant.id.hex, AccessType.ASSISTANT_WRITE):
|
||||
return AssistantNotExistsError.return_resp()
|
||||
if not user_payload.access_check(assistant.user_id, assistant.id, AccessType.ASSISTANT_WRITE):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
# 已上线不允许改动
|
||||
if assistant.status == AssistantStatus.ONLINE.value:
|
||||
@@ -464,7 +564,7 @@ class AssistantService(AssistantUtils):
|
||||
"""
|
||||
# 初始化agent
|
||||
auto_agent = AssistantAgent(assistant, '')
|
||||
await auto_agent.init_llm()
|
||||
await auto_agent.init_auto_update_llm()
|
||||
|
||||
# 自动生成描述
|
||||
assistant.desc = auto_agent.generate_description(assistant.prompt)
|
||||
|
||||
@@ -1,21 +1,10 @@
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Dict, List
|
||||
from uuid import UUID
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import httpx
|
||||
from bisheng_langchain.gpts.tools.api_tools.openapi import OpenApiTools
|
||||
|
||||
from bisheng.api.services.assistant_base import AssistantUtils
|
||||
from bisheng.api.services.knowledge_imp import decide_vectorstores
|
||||
from bisheng.api.services.openapi import OpenApiSchema
|
||||
from bisheng.api.utils import build_flow_no_yield
|
||||
from bisheng.api.v1.schemas import InputRequest
|
||||
from bisheng.database.models.assistant import Assistant, AssistantLink, AssistantLinkDao
|
||||
from bisheng.database.models.flow import FlowDao, FlowStatus
|
||||
from bisheng.database.models.gpts_tools import GptsTools, GptsToolsDao, GptsToolsType, AuthMethod
|
||||
from bisheng.database.models.knowledge import KnowledgeDao, Knowledge
|
||||
from bisheng_langchain.gpts.assistant import ConfigurableAssistant
|
||||
from bisheng_langchain.gpts.auto_optimization import (generate_breif_description,
|
||||
generate_opening_dialog,
|
||||
@@ -23,15 +12,31 @@ from bisheng_langchain.gpts.auto_optimization import (generate_breif_description
|
||||
from bisheng_langchain.gpts.auto_tool_selected import ToolInfo, ToolSelector
|
||||
from bisheng_langchain.gpts.load_tools import load_tools
|
||||
from bisheng_langchain.gpts.prompts import ASSISTANT_PROMPT_OPT
|
||||
from bisheng_langchain.gpts.utils import import_by_type, import_class
|
||||
from bisheng_langchain.gpts.tools.api_tools.openapi import OpenApiTools
|
||||
from langchain_core.callbacks import Callbacks
|
||||
from langchain_core.language_models import BaseLanguageModel
|
||||
from langchain_core.messages import AIMessage, HumanMessage
|
||||
from langchain_core.messages import AIMessage, HumanMessage, BaseMessage
|
||||
from langchain_core.runnables import RunnableConfig
|
||||
from langchain_core.tools import BaseTool, Tool
|
||||
from langchain_core.utils.function_calling import format_tool_to_openai_tool
|
||||
from langchain_core.vectorstores import VectorStoreRetriever
|
||||
from langgraph.prebuilt import create_react_agent
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.interface.embeddings.custom import FakeEmbedding
|
||||
from bisheng.api.services.assistant_base import AssistantUtils
|
||||
from bisheng.api.services.knowledge_imp import decide_vectorstores
|
||||
from bisheng.api.services.llm import LLMService
|
||||
from bisheng.api.services.openapi import OpenApiSchema
|
||||
from bisheng.api.utils import build_flow_no_yield
|
||||
from bisheng.api.v1.schemas import InputRequest
|
||||
from bisheng.database.constants import ToolPresetType
|
||||
from bisheng.database.models.assistant import Assistant, AssistantLink, AssistantLinkDao
|
||||
from bisheng.database.models.flow import FlowDao, FlowStatus
|
||||
from bisheng.database.models.gpts_tools import GptsTools, GptsToolsDao, GptsToolsType
|
||||
from bisheng.database.models.knowledge import Knowledge, KnowledgeDao
|
||||
from bisheng.mcp_manage.langchain.tool import McpTool
|
||||
from bisheng.mcp_manage.manager import ClientManager
|
||||
from bisheng.settings import settings
|
||||
from bisheng.utils.embedding import decide_embeddings
|
||||
|
||||
|
||||
@@ -50,9 +55,9 @@ class AssistantAgent(AssistantUtils):
|
||||
Finally, Write 'Grounded answer:' followed by a response to the user's last input in high quality natural english. Use the symbols <co: doc> and </co: doc> to indicate when a fact comes from a document in the search result, e.g <co: 4>my fact</co: 4> for a fact from document 4.
|
||||
|
||||
Additional instructions to note:
|
||||
- If the user's question is in Chinese, please answer it in Chinese.
|
||||
- If the user's question is in Chinese, please answer it in Chinese.
|
||||
- 当问题中有涉及到时间信息时,比如最近6个月、昨天、去年等,你需要用时间工具查询时间信息。
|
||||
"""
|
||||
""" # noqa
|
||||
|
||||
def __init__(self, assistant_info: Assistant, chat_id: str):
|
||||
self.assistant = assistant_info
|
||||
@@ -60,15 +65,17 @@ class AssistantAgent(AssistantUtils):
|
||||
self.tools: List[BaseTool] = []
|
||||
self.offline_flows = []
|
||||
self.agent: ConfigurableAssistant | None = None
|
||||
self.agent_executor_dict = {
|
||||
'ReAct': 'get_react_agent_executor',
|
||||
'function call': 'get_openai_functions_agent_executor',
|
||||
}
|
||||
self.current_agent_executor = None
|
||||
self.llm: BaseLanguageModel | None = None
|
||||
self.llm_agent_executor = None
|
||||
self.knowledge_skill_path = str(Path(__file__).parent / 'knowledge_skill.json')
|
||||
self.knowledge_skill_data = None
|
||||
# 知识库检索相关参数
|
||||
self.knowledge_retrive = {
|
||||
"max_content": 15000,
|
||||
"sort_by_source_and_index": False
|
||||
}
|
||||
self.knowledge_retriever = {'max_content': 15000, 'sort_by_source_and_index': False}
|
||||
|
||||
async def init_assistant(self, callbacks: Callbacks = None):
|
||||
await self.init_llm()
|
||||
@@ -76,85 +83,98 @@ class AssistantAgent(AssistantUtils):
|
||||
await self.init_agent()
|
||||
|
||||
async def init_llm(self):
|
||||
llm_params = self.get_llm_conf(self.assistant.model_name)
|
||||
if not llm_params:
|
||||
logger.error(
|
||||
f'act=init_llm llm_params is None, model_name: {self.assistant.model_name}')
|
||||
raise Exception(
|
||||
f'act=init_llm llm_params is None, model_name: {self.assistant.model_name}')
|
||||
# 获取配置的助手模型列表
|
||||
assistant_llm = LLMService.get_assistant_llm()
|
||||
if not assistant_llm.llm_list:
|
||||
raise Exception('助手推理模型列表为空')
|
||||
default_llm = None
|
||||
for one in assistant_llm.llm_list:
|
||||
if str(one.model_id) == self.assistant.model_name:
|
||||
default_llm = one
|
||||
break
|
||||
elif not default_llm and one.default:
|
||||
default_llm = one
|
||||
if not default_llm:
|
||||
raise Exception('未配置助手推理模型')
|
||||
|
||||
# 使用助手配置的 temperature
|
||||
llm_params['temperature'] = self.assistant.temperature
|
||||
self.llm_agent_executor = default_llm.agent_executor_type
|
||||
self.knowledge_retriever = {
|
||||
'max_content': default_llm.knowledge_max_content,
|
||||
'sort_by_source_and_index': default_llm.knowledge_sort_index
|
||||
}
|
||||
|
||||
if llm_params.get('agent_executor_type'):
|
||||
self.llm_agent_executor = llm_params.pop('agent_executor_type')
|
||||
# 初始化llm
|
||||
self.llm = LLMService.get_bisheng_llm(model_id=default_llm.model_id,
|
||||
temperature=self.assistant.temperature,
|
||||
streaming=default_llm.streaming)
|
||||
|
||||
# 如果模型有单独配置知识库检索参数,则使用模型配置的
|
||||
if llm_params.get('knowledge_retrive'):
|
||||
self.knowledge_retrive = llm_params.pop('knowledge_retrive')
|
||||
async def init_auto_update_llm(self):
|
||||
""" 初始化自动优化prompt等信息的llm实例 """
|
||||
assistant_llm = LLMService.get_assistant_llm()
|
||||
if not assistant_llm.auto_llm:
|
||||
raise Exception('未配置助手画像自动优化模型')
|
||||
|
||||
if llm_params['type'] == 'ChatOpenAI':
|
||||
llm_object = import_class('langchain_openai.ChatOpenAI')
|
||||
llm_params.pop('type')
|
||||
llm_params['model'] = llm_params.pop('model_name')
|
||||
if 'openai_proxy' in llm_params:
|
||||
openai_proxy = llm_params.pop('openai_proxy')
|
||||
llm_params['http_client'] = httpx.Client(proxies=openai_proxy)
|
||||
llm_params['http_async_client'] = httpx.AsyncClient(proxies=openai_proxy)
|
||||
self.llm = llm_object(**llm_params)
|
||||
else:
|
||||
llm_object = import_by_type(_type='llms', name=llm_params['type'])
|
||||
llm_params.pop('type')
|
||||
self.llm = llm_object(**llm_params)
|
||||
self.llm = LLMService.get_bisheng_llm(model_id=assistant_llm.auto_llm.model_id,
|
||||
temperature=self.assistant.temperature,
|
||||
streaming=assistant_llm.auto_llm.streaming)
|
||||
|
||||
async def get_knowledge_skill_data(self):
|
||||
if self.knowledge_skill_data:
|
||||
return self.knowledge_skill_data
|
||||
|
||||
with open(self.knowledge_skill_path, 'r', encoding='utf-8') as f:
|
||||
data = json.load(f)
|
||||
self.knowledge_skill_data = data
|
||||
return data
|
||||
|
||||
def parse_tool_params(self, tool: GptsTools) -> Dict:
|
||||
@staticmethod
|
||||
def parse_tool_params(tool: GptsTools) -> Dict:
|
||||
"""
|
||||
解析预置工具的初始化参数
|
||||
"""
|
||||
# 特殊处理下bisheng_code_interpreter的参数
|
||||
if tool.tool_key == 'bisheng_code_interpreter':
|
||||
return {'minio': settings.get_minio_conf().model_dump()}
|
||||
if not tool.extra:
|
||||
return {}
|
||||
params = json.loads(tool.extra)
|
||||
|
||||
# 判断是否需要从系统配置里获取, 不需要从系统配置获取则用本身配置的
|
||||
if params.get('&initdb_conf_key'):
|
||||
return self.get_initdb_conf_by_more_key(params.get('&initdb_conf_key'))
|
||||
return params
|
||||
|
||||
@staticmethod
|
||||
def sync_init_preset_tools(tool_list: List[GptsTools],
|
||||
llm: BaseLanguageModel = None,
|
||||
callbacks: Callbacks = None):
|
||||
"""
|
||||
初始化预置工具列表
|
||||
"""
|
||||
tool_name_param = {
|
||||
tool.tool_key: AssistantAgent.parse_tool_params(tool)
|
||||
for tool in tool_list
|
||||
}
|
||||
tool_langchain = load_tools(tool_params=tool_name_param, llm=llm, callbacks=callbacks)
|
||||
return tool_langchain
|
||||
|
||||
async def init_preset_tools(self, tool_list: List[GptsTools], callbacks: Callbacks = None):
|
||||
"""
|
||||
初始化预置工具列表
|
||||
"""
|
||||
tool_name_param = {
|
||||
tool.tool_key: self.parse_tool_params(tool)
|
||||
tool.tool_key: AssistantAgent.parse_tool_params(tool)
|
||||
for tool in tool_list
|
||||
}
|
||||
tool_langchain = load_tools(tool_params=tool_name_param,
|
||||
llm=self.llm,
|
||||
callbacks=callbacks)
|
||||
tool_langchain = load_tools(tool_params=tool_name_param, llm=self.llm, callbacks=callbacks)
|
||||
return tool_langchain
|
||||
|
||||
@staticmethod
|
||||
async def parse_personal_params(tool: GptsTools, all_tool_type: Dict[int, GptsToolsType]) -> Dict:
|
||||
def parse_personal_params(tool: GptsTools, all_tool_type: Dict[int, GptsToolsType]) -> Dict:
|
||||
"""
|
||||
解析自定义工具的初始化参数
|
||||
"""
|
||||
tool_type_info = all_tool_type.get(tool.type)
|
||||
if not tool_type_info:
|
||||
raise Exception(f'获取工具类型失败,tool_type_id: {tool.type}')
|
||||
return OpenApiSchema.parse_openapi_tool_params(tool.name, tool.desc, tool.extra,
|
||||
tool_type_info.server_host, tool_type_info.auth_method,
|
||||
tool_type_info.auth_type, tool_type_info.api_key)
|
||||
extra_json = json.loads(tool.extra) if tool.extra else {}
|
||||
extra_json.update(json.loads(tool_type_info.extra) if tool_type_info.extra else {})
|
||||
return OpenApiSchema.parse_openapi_tool_params(tool.name, tool.desc, json.dumps(extra_json),
|
||||
tool_type_info.server_host,
|
||||
tool_type_info.auth_method,
|
||||
tool_type_info.auth_type,
|
||||
tool_type_info.api_key)
|
||||
|
||||
async def init_personal_tools(self, tool_list: List[GptsTools], callbacks: Callbacks = None):
|
||||
@staticmethod
|
||||
def sync_init_personal_tools(tool_list: List[GptsTools], callbacks: Callbacks = None):
|
||||
"""
|
||||
初始化自定义工具列表
|
||||
"""
|
||||
@@ -163,33 +183,156 @@ class AssistantAgent(AssistantUtils):
|
||||
all_tool_type = {one.id: one for one in all_tool_type}
|
||||
tool_langchain = []
|
||||
for one in tool_list:
|
||||
tool_params = await self.parse_personal_params(one, all_tool_type)
|
||||
tool_params = AssistantAgent.parse_personal_params(one, all_tool_type)
|
||||
openapi_tool = OpenApiTools.get_api_tool(one.tool_key, **tool_params)
|
||||
openapi_tool.callbacks = callbacks
|
||||
tool_langchain.append(openapi_tool)
|
||||
return tool_langchain
|
||||
|
||||
async def init_knowledge_tool(self, knowledge: Knowledge, callbacks: Callbacks = None):
|
||||
@staticmethod
|
||||
def sync_init_mcp_tools(tool_list: List[GptsTools], callbacks: Callbacks = None):
|
||||
"""
|
||||
初始化mcp工具列表
|
||||
"""
|
||||
tool_type_ids = [one.type for one in tool_list]
|
||||
all_tool_type = GptsToolsDao.get_all_tool_type(tool_type_ids)
|
||||
all_tool_type = {one.id: one for one in all_tool_type}
|
||||
tool_langchain = []
|
||||
for one in tool_list:
|
||||
tool_type = all_tool_type.get(one.type)
|
||||
input_schema = json.loads(one.extra)
|
||||
mcp_client = ClientManager.sync_connect_mcp_from_json(tool_type.openapi_schema)
|
||||
mcp_tool = McpTool.get_mcp_tool(name=one.tool_key, description=one.desc, mcp_client=mcp_client,
|
||||
mcp_tool_name=one.name, arg_schema=input_schema['inputSchema'],
|
||||
callbacks=callbacks)
|
||||
tool_langchain.append(mcp_tool)
|
||||
return tool_langchain
|
||||
|
||||
@staticmethod
|
||||
async def async_init_mcp_tools(tool_list: List[GptsTools], callbacks: Callbacks = None):
|
||||
"""
|
||||
初始化mcp工具列表
|
||||
"""
|
||||
tool_type_ids = [one.type for one in tool_list]
|
||||
all_tool_type = GptsToolsDao.get_all_tool_type(tool_type_ids)
|
||||
all_tool_type = {one.id: one for one in all_tool_type}
|
||||
tool_langchain = []
|
||||
for one in tool_list:
|
||||
tool_type = all_tool_type.get(one.type)
|
||||
input_schema = json.loads(one.extra)
|
||||
mcp_client = await ClientManager.connect_mcp_from_json(tool_type.openapi_schema)
|
||||
mcp_tool = McpTool.get_mcp_tool(name=one.tool_key, description=one.desc, mcp_client=mcp_client,
|
||||
mcp_tool_name=one.name, arg_schema=input_schema['inputSchema'],
|
||||
callbacks=callbacks)
|
||||
tool_langchain.append(mcp_tool)
|
||||
return tool_langchain
|
||||
|
||||
@staticmethod
|
||||
def sync_init_knowledge_tool(knowledge: Knowledge,
|
||||
llm: BaseLanguageModel,
|
||||
callbacks: Callbacks = None,
|
||||
knowledge_retriever: dict = None):
|
||||
"""
|
||||
初始化知识库工具
|
||||
"""
|
||||
embeddings = decide_embeddings(knowledge.model)
|
||||
search_kwargs = {}
|
||||
vector_client = decide_vectorstores(knowledge.collection_name, 'Milvus', embeddings)
|
||||
es_vector_client = decide_vectorstores(knowledge.index_name, 'ElasticKeywordsSearch', embeddings)
|
||||
if isinstance(vector_client, VectorStoreRetriever):
|
||||
vector_client = vector_client.vectorstore
|
||||
vector_client.partition_key = knowledge.id
|
||||
|
||||
es_vector_client = decide_vectorstores(knowledge.index_name, 'ElasticKeywordsSearch',
|
||||
embeddings)
|
||||
tool_params = {
|
||||
"bisheng_rag": {
|
||||
"name": f"knowledge_{knowledge.id}",
|
||||
"description": f"{knowledge.name}:{knowledge.description}",
|
||||
"vector_store": vector_client,
|
||||
"keyword_store": es_vector_client,
|
||||
"llm": self.llm
|
||||
'bisheng_rag': {
|
||||
'name': f'knowledge_{knowledge.id}',
|
||||
'description': f'{knowledge.name}:{knowledge.description}',
|
||||
'vector_store': vector_client,
|
||||
'keyword_store': es_vector_client,
|
||||
'llm': llm
|
||||
}
|
||||
}
|
||||
tool_params['bisheng_rag'].update(self.knowledge_retrive)
|
||||
tool = load_tools(tool_params=tool_params, llm=self.llm, callbacks=callbacks)
|
||||
if knowledge_retriever:
|
||||
tool_params['bisheng_rag'].update(knowledge_retriever)
|
||||
tool = load_tools(tool_params=tool_params, llm=llm, callbacks=callbacks)
|
||||
return tool
|
||||
|
||||
async def init_knowledge_tool(self, knowledge: Knowledge, callbacks: Callbacks = None):
|
||||
"""
|
||||
初始化知识库工具
|
||||
"""
|
||||
return self.sync_init_knowledge_tool(knowledge,
|
||||
self.llm,
|
||||
callbacks,
|
||||
self.knowledge_retriever)
|
||||
|
||||
@staticmethod
|
||||
def parse_tools_type(tool_ids: List[int]) -> (list, list, list):
|
||||
"""
|
||||
解析工具类型
|
||||
"""
|
||||
tools_model: List[GptsTools] = GptsToolsDao.get_list_by_ids(tool_ids)
|
||||
preset_tools = []
|
||||
personal_tools = []
|
||||
mcp_tools = []
|
||||
for one in tools_model:
|
||||
if one.is_preset == ToolPresetType.PRESET.value:
|
||||
preset_tools.append(one)
|
||||
elif one.is_preset == ToolPresetType.API.value:
|
||||
personal_tools.append(one)
|
||||
else:
|
||||
mcp_tools.append(one)
|
||||
return preset_tools, personal_tools, mcp_tools
|
||||
|
||||
@staticmethod
|
||||
def init_tools_by_toolid(
|
||||
tool_ids: List[int],
|
||||
llm: BaseLanguageModel,
|
||||
callbacks: Callbacks = None,
|
||||
):
|
||||
""" 通过id初始化tool !!! 只能在没有事件循环的线程中调用 """
|
||||
tools = []
|
||||
preset_tools, personal_tools, mcp_tools = AssistantAgent.parse_tools_type(tool_ids)
|
||||
if preset_tools:
|
||||
tool_langchain = AssistantAgent.sync_init_preset_tools(preset_tools, llm, callbacks)
|
||||
logger.info('act=build_preset_tools size={} return_tools={}', len(preset_tools),
|
||||
len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
if personal_tools:
|
||||
tool_langchain = AssistantAgent.sync_init_personal_tools(personal_tools, callbacks)
|
||||
logger.info('act=build_personal_tools size={} return_tools={}', len(personal_tools),
|
||||
len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
if mcp_tools:
|
||||
tool_langchain = AssistantAgent.sync_init_mcp_tools(mcp_tools, callbacks)
|
||||
logger.info('act=build_mcp_tools size={} return_tools={}', len(mcp_tools),
|
||||
len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
return tools
|
||||
|
||||
@staticmethod
|
||||
async def init_tools_by_tool_ids(tool_ids: List[int],
|
||||
llm: BaseLanguageModel,
|
||||
callbacks: Callbacks = None, ):
|
||||
tools = []
|
||||
preset_tools, personal_tools, mcp_tools = AssistantAgent.parse_tools_type(tool_ids)
|
||||
if preset_tools:
|
||||
tool_langchain = AssistantAgent.sync_init_preset_tools(preset_tools, llm, callbacks)
|
||||
logger.info('act=build_preset_tools size={} return_tools={}', len(preset_tools),
|
||||
len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
if personal_tools:
|
||||
tool_langchain = AssistantAgent.sync_init_personal_tools(personal_tools, callbacks)
|
||||
logger.info('act=build_personal_tools size={} return_tools={}', len(personal_tools),
|
||||
len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
if mcp_tools:
|
||||
tools_langchain = await AssistantAgent.async_init_mcp_tools(mcp_tools, callbacks)
|
||||
logger.info('act=build_mcp_tools size={} return_tools={}', len(mcp_tools),
|
||||
len(tools_langchain))
|
||||
tools += tools_langchain
|
||||
return tools
|
||||
|
||||
async def init_tools(self, callbacks: Callbacks = None):
|
||||
"""通过名称获取tool 列表
|
||||
tools_name_param:: {name: params}
|
||||
@@ -206,23 +349,7 @@ class AssistantAgent(AssistantUtils):
|
||||
else:
|
||||
flow_links.append(link)
|
||||
if tool_ids:
|
||||
tools_model: List[GptsTools] = GptsToolsDao.get_list_by_ids(tool_ids)
|
||||
preset_tools = []
|
||||
personal_tools = []
|
||||
for one in tools_model:
|
||||
if one.is_preset:
|
||||
preset_tools.append(one)
|
||||
else:
|
||||
personal_tools.append(one)
|
||||
if preset_tools:
|
||||
tool_langchain = await self.init_preset_tools(preset_tools, callbacks)
|
||||
logger.info('act=build_preset_tools size={} return_tools={}', len(preset_tools), len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
if personal_tools:
|
||||
tool_langchain = await self.init_personal_tools(personal_tools, callbacks)
|
||||
logger.info('act=build_personal_tools size={} return_tools={}', len(personal_tools),
|
||||
len(tool_langchain))
|
||||
tools += tool_langchain
|
||||
tools = await self.init_tools_by_tool_ids(tool_ids, self.llm, callbacks)
|
||||
|
||||
# flow + knowledge
|
||||
flow_data = FlowDao.get_flow_by_ids([link.flow_id for link in flow_links if link.flow_id])
|
||||
@@ -234,11 +361,12 @@ class AssistantAgent(AssistantUtils):
|
||||
for link in flow_links:
|
||||
knowledge_id = link.knowledge_id
|
||||
if knowledge_id:
|
||||
knowledge_tool = await self.init_knowledge_tool(knowledge_data[knowledge_id], callbacks)
|
||||
knowledge_tool = await self.init_knowledge_tool(knowledge_data[knowledge_id],
|
||||
callbacks)
|
||||
tools.extend(knowledge_tool)
|
||||
else:
|
||||
tmp_flow_id = UUID(link.flow_id).hex
|
||||
one_flow_data = flow_id2data.get(UUID(link.flow_id))
|
||||
tmp_flow_id = link.flow_id
|
||||
one_flow_data = flow_id2data.get(link.flow_id)
|
||||
tool_name = f'flow_{link.flow_id}'
|
||||
if not one_flow_data:
|
||||
logger.warning('act=init_tools not find flow_id: {}', link.flow_id)
|
||||
@@ -256,7 +384,7 @@ class AssistantAgent(AssistantUtils):
|
||||
artifacts=artifacts,
|
||||
process_file=True,
|
||||
flow_id=tmp_flow_id,
|
||||
chat_id=self.assistant.id.hex)
|
||||
chat_id=self.assistant.id)
|
||||
built_object = await graph.abuild()
|
||||
logger.info('act=init_flow_tool build_end')
|
||||
flow_tool = Tool(name=tool_name,
|
||||
@@ -276,19 +404,23 @@ class AssistantAgent(AssistantUtils):
|
||||
初始化智能体的agent
|
||||
"""
|
||||
# 引入agent执行参数
|
||||
agent_executor_params = self.get_agent_executor()
|
||||
agent_executor_type = self.llm_agent_executor or agent_executor_params.pop('type')
|
||||
agent_executor_type = self.llm_agent_executor
|
||||
self.current_agent_executor = agent_executor_type
|
||||
# 做转换
|
||||
agent_executor_type = self.agent_executor_dict.get(agent_executor_type,
|
||||
agent_executor_type)
|
||||
|
||||
prompt = self.assistant.prompt
|
||||
if self.assistant.model_name.startswith("command-r"):
|
||||
if getattr(self.llm, 'model_name', '').startswith('command-r'):
|
||||
prompt = self.ASSISTANT_PROMPT_COHERE.format(preamble=prompt)
|
||||
|
||||
# 初始化agent
|
||||
self.agent = ConfigurableAssistant(agent_executor_type=agent_executor_type,
|
||||
tools=self.tools,
|
||||
llm=self.llm,
|
||||
assistant_message=prompt,
|
||||
**agent_executor_params)
|
||||
if self.current_agent_executor == 'ReAct':
|
||||
# 初始化agent
|
||||
self.agent = ConfigurableAssistant(agent_executor_type=agent_executor_type,
|
||||
tools=self.tools,
|
||||
llm=self.llm,
|
||||
assistant_message=prompt)
|
||||
else:
|
||||
self.agent = create_react_agent(self.llm, self.tools, prompt=prompt, checkpointer=False)
|
||||
|
||||
async def optimize_assistant_prompt(self):
|
||||
""" 自动优化生成prompt """
|
||||
@@ -327,25 +459,102 @@ class AssistantAgent(AssistantUtils):
|
||||
tool_selector = ToolSelector(llm=self.llm, tools=tool_list)
|
||||
return tool_selector.select(self.assistant.name, prompt)
|
||||
|
||||
async def run(self, query: str, chat_history: List = None, callback: Callbacks = None):
|
||||
async def fake_callback(self, callback: Callbacks):
|
||||
if not callback:
|
||||
return
|
||||
# 假回调,将已下线的技能回调给前端
|
||||
for one in self.offline_flows:
|
||||
run_id = uuid.uuid4()
|
||||
await callback[0].on_tool_start({
|
||||
'name': one,
|
||||
},
|
||||
input_str='flow is offline',
|
||||
run_id=run_id)
|
||||
await callback[0].on_tool_end(output='flow is offline', name=one, run_id=run_id)
|
||||
|
||||
async def record_chat_history(self, message: List[Any]):
|
||||
# 记录助手的聊天历史
|
||||
if not os.getenv('BISHENG_RECORD_HISTORY'):
|
||||
return
|
||||
try:
|
||||
os.makedirs('/app/data/history', exist_ok=True)
|
||||
with open(f'/app/data/history/{self.assistant.id}_{time.time()}.json',
|
||||
'w',
|
||||
encoding='utf-8') as f:
|
||||
json.dump(
|
||||
{
|
||||
'system': self.assistant.prompt,
|
||||
'message': message,
|
||||
'tools': [format_tool_to_openai_tool(t) for t in self.tools]
|
||||
},
|
||||
f,
|
||||
ensure_ascii=False)
|
||||
except Exception as e:
|
||||
logger.error(f'record assistant history error: {str(e)}')
|
||||
|
||||
async def trim_messages(self, messages: List[Any]) -> List[Any]:
|
||||
# 获取encoding
|
||||
enc = self.cl100k_base()
|
||||
|
||||
def get_finally_message(new_messages: List[Any]) -> List[Any]:
|
||||
# 修剪到只有一条记录则不再处理
|
||||
if len(new_messages) == 1:
|
||||
return new_messages
|
||||
total_count = 0
|
||||
for one in new_messages:
|
||||
if isinstance(one, HumanMessage):
|
||||
total_count += len(enc.encode(one.content))
|
||||
elif isinstance(one, AIMessage):
|
||||
total_count += len(enc.encode(one.content))
|
||||
if 'tool_calls' in one.additional_kwargs:
|
||||
total_count += len(
|
||||
enc.encode(json.dumps(one.additional_kwargs['tool_calls'], ensure_ascii=False))
|
||||
)
|
||||
else:
|
||||
total_count += len(enc.encode(str(one.content)))
|
||||
if total_count > self.assistant.max_token:
|
||||
return get_finally_message(new_messages[1:])
|
||||
return new_messages
|
||||
|
||||
return get_finally_message(messages)
|
||||
|
||||
async def run(self, query: str, chat_history: List = None, callback: Callbacks = None) -> List[BaseMessage]:
|
||||
"""
|
||||
运行智能体对话
|
||||
"""
|
||||
await self.fake_callback(callback)
|
||||
|
||||
if chat_history:
|
||||
chat_history.append(HumanMessage(content=query))
|
||||
inputs = chat_history
|
||||
else:
|
||||
inputs = [HumanMessage(content=query)]
|
||||
|
||||
# 假回调,将已下线的技能回调给前端
|
||||
for one in self.offline_flows:
|
||||
if callback is not None:
|
||||
run_id = uuid.uuid4()
|
||||
await callback[0].on_tool_start({
|
||||
'name': one,
|
||||
}, input_str='', run_id=run_id)
|
||||
await callback[0].on_tool_end(output='', name=one, run_id=run_id)
|
||||
result = await self.agent.ainvoke(inputs, config=RunnableConfig(callbacks=callback))
|
||||
# 包含了history,将history排除, 默认取最后一个为最终结果
|
||||
res = [result[-1]]
|
||||
return res
|
||||
# trim message
|
||||
inputs = await self.trim_messages(inputs)
|
||||
|
||||
if self.current_agent_executor == 'ReAct':
|
||||
result = await self.react_run(inputs, callback)
|
||||
else:
|
||||
result = await self.agent.ainvoke({'messages': inputs}, config=RunnableConfig(callbacks=callback))
|
||||
result = result['messages']
|
||||
|
||||
# 记录聊天历史
|
||||
await self.record_chat_history([one.to_json() for one in result])
|
||||
|
||||
return result
|
||||
|
||||
async def react_run(self, inputs: List, callback: Callbacks = None):
|
||||
""" react 模式的输入和执行 """
|
||||
result = await self.agent.ainvoke({
|
||||
'input': inputs[-1].content,
|
||||
'chat_history': inputs[:-1],
|
||||
}, config=RunnableConfig(callbacks=callback))
|
||||
logger.debug(f"react_run result: {result}")
|
||||
output = result['agent_outcome'].return_values['output']
|
||||
if isinstance(output, dict):
|
||||
output = list(output.values())[0]
|
||||
for one in result['intermediate_steps']:
|
||||
inputs.append(one[0])
|
||||
inputs.append(AIMessage(content=output))
|
||||
return inputs
|
||||
|
||||
@@ -1,46 +1,37 @@
|
||||
from typing import Dict
|
||||
import os
|
||||
|
||||
from bisheng.settings import settings
|
||||
from tiktoken.load import load_tiktoken_bpe
|
||||
from tiktoken.core import Encoding as TikTokenEncoding
|
||||
|
||||
|
||||
class AssistantUtils:
|
||||
# 忽略助手配置已从系统配置中移除,暂不需要此类的方法
|
||||
|
||||
@classmethod
|
||||
def get_gpts_conf(cls, key=None):
|
||||
gpts_conf = settings.get_from_db('gpts')
|
||||
if key:
|
||||
return gpts_conf.get(key)
|
||||
return gpts_conf
|
||||
@staticmethod
|
||||
def cl100k_base() -> TikTokenEncoding:
|
||||
ENDOFTEXT = "<|endoftext|>"
|
||||
FIM_PREFIX = "<|fim_prefix|>"
|
||||
FIM_MIDDLE = "<|fim_middle|>"
|
||||
FIM_SUFFIX = "<|fim_suffix|>"
|
||||
ENDOFPROMPT = "<|endofprompt|>"
|
||||
|
||||
@classmethod
|
||||
def get_llm_conf(cls, llm_name: str) -> dict:
|
||||
llm_list = cls.get_gpts_conf('llms')
|
||||
for one in llm_list:
|
||||
if one['model_name'] == llm_name:
|
||||
return one
|
||||
return llm_list[0]
|
||||
tiktoken_file = os.path.join(os.path.dirname(__file__), "tiktoken_file/cl100k_base.tiktoken")
|
||||
|
||||
@classmethod
|
||||
def get_prompt_type(cls):
|
||||
return cls.get_gpts_conf('prompt_type')
|
||||
|
||||
@classmethod
|
||||
def get_agent_executor(cls):
|
||||
return cls.get_gpts_conf('agent_executor')
|
||||
|
||||
@classmethod
|
||||
def get_default_retrieval(cls) -> str:
|
||||
return cls.get_gpts_conf('default-retrieval')
|
||||
|
||||
@classmethod
|
||||
def get_initdb_conf_by_more_key(cls, key: str) -> Dict:
|
||||
"""
|
||||
根据多层级的key,获取对应的配置。
|
||||
:param key: 例如:gpts.tools.code_interpreter 表示获取 gpts['tools']['code_interpreter']的内容
|
||||
"""
|
||||
# 因为属于系统配置级别,不做不存在的判断。不存在直接抛出异常
|
||||
key_list = key.split('.')
|
||||
root_conf = settings.get_from_db(key_list[0].strip())
|
||||
for one in key_list[1:]:
|
||||
root_conf = root_conf[one.strip()]
|
||||
return root_conf
|
||||
mergeable_ranks = load_tiktoken_bpe(
|
||||
# "https://openaipublic.blob.core.windows.net/encodings/cl100k_base.tiktoken",
|
||||
tiktoken_file,
|
||||
expected_hash="223921b76ee99bde995b7ff738513eef100fb51d18c93597a113bcffe865b2a7",
|
||||
)
|
||||
special_tokens = {
|
||||
ENDOFTEXT: 100257,
|
||||
FIM_PREFIX: 100258,
|
||||
FIM_MIDDLE: 100259,
|
||||
FIM_SUFFIX: 100260,
|
||||
ENDOFPROMPT: 100276,
|
||||
}
|
||||
return TikTokenEncoding(**{
|
||||
"name": "cl100k_base",
|
||||
"pat_str": r"""'(?i:[sdmt]|ll|ve|re)|[^\r\n\p{L}\p{N}]?+\p{L}++|\p{N}{1,3}+| ?[^\s\p{L}\p{N}]++[\r\n]*+|\s++$|\s*[\r\n]|\s+(?!\S)|\s""",
|
||||
"mergeable_ranks": mergeable_ranks,
|
||||
"special_tokens": special_tokens,
|
||||
})
|
||||
|
||||
@@ -0,0 +1,565 @@
|
||||
import csv
|
||||
from datetime import datetime
|
||||
from tempfile import NamedTemporaryFile
|
||||
from typing import Any, List, Optional
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.errcode.base import UnAuthorizedError
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.v1.schema.chat_schema import AppChatList
|
||||
from bisheng.api.v1.schema.workflow import WorkflowEventType
|
||||
from bisheng.api.v1.schemas import resp_200
|
||||
from bisheng.database.models.assistant import AssistantDao, Assistant
|
||||
from bisheng.database.models.audit_log import AuditLog, SystemId, EventType, ObjectType, AuditLogDao
|
||||
from bisheng.database.models.flow import FlowDao, Flow, FlowType
|
||||
from bisheng.database.models.group import Group
|
||||
from bisheng.database.models.group_resource import GroupResourceDao, ResourceTypeEnum
|
||||
from bisheng.database.models.knowledge import KnowledgeDao, Knowledge
|
||||
from bisheng.database.models.message import ChatMessageDao, LikedType
|
||||
from bisheng.database.models.role import Role
|
||||
from bisheng.database.models.session import MessageSessionDao, SensitiveStatus
|
||||
from bisheng.database.models.user import UserDao, User
|
||||
from bisheng.database.models.user_group import UserGroupDao
|
||||
from bisheng.settings import settings
|
||||
from bisheng.utils import generate_uuid
|
||||
from bisheng.utils.minio_client import MinioClient
|
||||
|
||||
|
||||
class AuditLogService:
|
||||
|
||||
@classmethod
|
||||
def get_audit_log(cls, login_user: UserPayload, group_ids, operator_ids, start_time, end_time,
|
||||
system_id, event_type, page, limit) -> Any:
|
||||
groups = group_ids
|
||||
if not login_user.is_admin():
|
||||
groups = [str(one.group_id) for one in UserGroupDao.get_user_admin_group(login_user.user_id)]
|
||||
# 不是任何用戶组的管理员
|
||||
if not groups:
|
||||
return UnAuthorizedError.return_resp()
|
||||
# 将筛选条件的group_id和管理员有权限的groups做交集
|
||||
if group_ids:
|
||||
groups = list(set(groups) & set(group_ids))
|
||||
if not groups:
|
||||
return UnAuthorizedError.return_resp()
|
||||
data, total = AuditLogDao.get_audit_logs(groups, operator_ids, start_time, end_time, system_id, event_type,
|
||||
page, limit)
|
||||
return resp_200(data={'data': data, 'total': total})
|
||||
|
||||
@classmethod
|
||||
def get_all_operators(cls, login_user: UserPayload) -> Any:
|
||||
groups = []
|
||||
if not login_user.is_admin():
|
||||
groups = [one.group_id for one in UserGroupDao.get_user_admin_group(login_user.user_id)]
|
||||
|
||||
data = AuditLogDao.get_all_operators(groups)
|
||||
res = {}
|
||||
for one in data:
|
||||
if not one[1]:
|
||||
continue
|
||||
res[one[0]] = {'user_id': one[0], 'user_name': one[1]}
|
||||
return resp_200(data=list(res.values()))
|
||||
|
||||
@classmethod
|
||||
def _chat_log(cls, user: UserPayload, ip_address: str, event_type: EventType, object_type: ObjectType,
|
||||
object_id: str, object_name: str, resource_type: ResourceTypeEnum):
|
||||
# 获取资源所属的分组
|
||||
groups = GroupResourceDao.get_resource_group(resource_type, object_id)
|
||||
group_ids = [one.group_id for one in groups]
|
||||
audit_log = AuditLog(
|
||||
operator_id=user.user_id,
|
||||
operator_name=user.user_name,
|
||||
group_ids=group_ids,
|
||||
system_id=SystemId.CHAT.value,
|
||||
event_type=event_type.value,
|
||||
object_type=object_type.value,
|
||||
object_id=object_id,
|
||||
object_name=object_name,
|
||||
ip_address=ip_address,
|
||||
)
|
||||
AuditLogDao.insert_audit_logs([audit_log])
|
||||
|
||||
@classmethod
|
||||
def create_chat_assistant(cls, user: UserPayload, ip_address: str, assistant_id: str):
|
||||
"""
|
||||
新建助手会话的审计日志
|
||||
"""
|
||||
logger.info(f"act=create_chat_assistant user={user.user_name} ip={ip_address} assistant={assistant_id}")
|
||||
# 获取助手详情
|
||||
assistant_info = AssistantDao.get_one_assistant(assistant_id)
|
||||
cls._chat_log(user, ip_address, EventType.CREATE_CHAT, ObjectType.ASSISTANT,
|
||||
assistant_id, assistant_info.name, ResourceTypeEnum.ASSISTANT)
|
||||
|
||||
@classmethod
|
||||
def create_chat_flow(cls, user: UserPayload, ip_address: str, flow_id: str, flow_info=None):
|
||||
"""
|
||||
新建技能会话的审计日志
|
||||
"""
|
||||
logger.info(f"act=create_chat_flow user={user.user_name} ip={ip_address} flow={flow_id}")
|
||||
if not flow_info:
|
||||
flow_info = FlowDao.get_flow_by_id(flow_id)
|
||||
cls._chat_log(user, ip_address, EventType.CREATE_CHAT, ObjectType.FLOW,
|
||||
flow_id, flow_info.name, ResourceTypeEnum.FLOW)
|
||||
|
||||
@classmethod
|
||||
def create_chat_workflow(cls, user: UserPayload, ip_address: str, flow_id: str, flow_info=None):
|
||||
"""
|
||||
新建工作流会话的审计日志
|
||||
"""
|
||||
logger.info(f"act=create_chat_workflow user={user.user_name} ip={ip_address} flow={flow_id}")
|
||||
if not flow_info:
|
||||
flow_info = FlowDao.get_flow_by_id(flow_id)
|
||||
cls._chat_log(user, ip_address, EventType.CREATE_CHAT, ObjectType.WORK_FLOW,
|
||||
flow_id, flow_info.name, ResourceTypeEnum.WORK_FLOW)
|
||||
|
||||
@classmethod
|
||||
def delete_chat_flow(cls, user: UserPayload, ip_address: str, flow_info: Flow):
|
||||
"""
|
||||
删除技能会话的审计日志
|
||||
"""
|
||||
logger.info(f"act=delete_chat_flow user={user.user_name} ip={ip_address} flow={flow_info.id}")
|
||||
cls._chat_log(user, ip_address, EventType.DELETE_CHAT, ObjectType.FLOW,
|
||||
flow_info.id, flow_info.name, ResourceTypeEnum.FLOW)
|
||||
|
||||
@classmethod
|
||||
def delete_chat_workflow(cls, user: UserPayload, ip_address: str, flow_info: Flow):
|
||||
"""
|
||||
删除技能会话的审计日志
|
||||
"""
|
||||
logger.info(f"act=delete_chat_workflow user={user.user_name} ip={ip_address} flow={flow_info.id}")
|
||||
cls._chat_log(user, ip_address, EventType.DELETE_CHAT, ObjectType.WORK_FLOW,
|
||||
flow_info.id, flow_info.name, ResourceTypeEnum.WORK_FLOW)
|
||||
|
||||
@classmethod
|
||||
def delete_chat_assistant(cls, user: UserPayload, ip_address: str, assistant_info: Assistant):
|
||||
"""
|
||||
删除助手会话的审计日志
|
||||
"""
|
||||
logger.info(f"act=delete_assistant_flow user={user.user_name} ip={ip_address} assistant={assistant_info.id}")
|
||||
cls._chat_log(user, ip_address, EventType.DELETE_CHAT, ObjectType.ASSISTANT,
|
||||
assistant_info.id, assistant_info.name, ResourceTypeEnum.ASSISTANT)
|
||||
|
||||
@classmethod
|
||||
def _build_log(cls, user: UserPayload, ip_address: str, event_type: EventType, object_type: ObjectType,
|
||||
object_id: str,
|
||||
object_name: str, resource_type: ResourceTypeEnum):
|
||||
"""
|
||||
构建模块的审计日志
|
||||
"""
|
||||
# 获取资源属于哪些用户组
|
||||
groups = GroupResourceDao.get_resource_group(resource_type, object_id)
|
||||
group_ids = [one.group_id for one in groups]
|
||||
|
||||
# 插入审计日志
|
||||
audit_log = AuditLog(
|
||||
operator_id=user.user_id,
|
||||
operator_name=user.user_name,
|
||||
group_ids=group_ids,
|
||||
system_id=SystemId.BUILD.value,
|
||||
event_type=event_type.value,
|
||||
object_type=object_type.value,
|
||||
object_id=object_id,
|
||||
object_name=object_name,
|
||||
ip_address=ip_address,
|
||||
)
|
||||
AuditLogDao.insert_audit_logs([audit_log])
|
||||
|
||||
@classmethod
|
||||
def create_build_flow(cls, user: UserPayload, ip_address: str, flow_id: str, flow_type: Optional[int] = None):
|
||||
"""
|
||||
新建技能的审计日志
|
||||
"""
|
||||
obj_type = ObjectType.FLOW
|
||||
rs_type = ResourceTypeEnum.FLOW
|
||||
if flow_type == FlowType.WORKFLOW.value:
|
||||
obj_type = ObjectType.WORK_FLOW
|
||||
rs_type = ResourceTypeEnum.WORK_FLOW
|
||||
logger.info(f"act=create_build_flow user={user.user_name} ip={ip_address} flow={flow_id}")
|
||||
flow_info = FlowDao.get_flow_by_id(flow_id)
|
||||
cls._build_log(user, ip_address, EventType.CREATE_BUILD, obj_type,
|
||||
flow_info.id, flow_info.name, rs_type)
|
||||
|
||||
@classmethod
|
||||
def update_build_flow(cls, user: UserPayload, ip_address: str, flow_id: str, flow_type: Optional[int] = None):
|
||||
"""
|
||||
更新技能的审计日志
|
||||
"""
|
||||
obj_type = ObjectType.FLOW
|
||||
rs_type = ResourceTypeEnum.FLOW
|
||||
if flow_type == FlowType.WORKFLOW.value:
|
||||
obj_type = ObjectType.WORK_FLOW
|
||||
rs_type = ResourceTypeEnum.WORK_FLOW
|
||||
logger.info(f"act=update_build_flow user={user.user_name} ip={ip_address} flow={flow_id}")
|
||||
flow_info = FlowDao.get_flow_by_id(flow_id)
|
||||
cls._build_log(user, ip_address, EventType.UPDATE_BUILD, obj_type,
|
||||
flow_info.id, flow_info.name, rs_type)
|
||||
|
||||
@classmethod
|
||||
def delete_build_flow(cls, user: UserPayload, ip_address: str, flow_info: Flow, flow_type: Optional[int] = None):
|
||||
"""
|
||||
删除技能的审计日志
|
||||
"""
|
||||
obj_type = ObjectType.FLOW
|
||||
rs_type = ResourceTypeEnum.FLOW
|
||||
if flow_type == FlowType.WORKFLOW.value:
|
||||
obj_type = ObjectType.WORK_FLOW
|
||||
rs_type = ResourceTypeEnum.WORK_FLOW
|
||||
logger.info(f"act=delete_build_flow user={user.user_name} ip={ip_address} flow={flow_info.id}")
|
||||
cls._build_log(user, ip_address, EventType.DELETE_BUILD, obj_type,
|
||||
flow_info.id, flow_info.name, rs_type)
|
||||
|
||||
@classmethod
|
||||
def create_build_assistant(cls, user: UserPayload, ip_address: str, assistant_id: str):
|
||||
"""
|
||||
新建助手的审计日志
|
||||
"""
|
||||
logger.info(f"act=create_build_assistant user={user.user_name} ip={ip_address} assistant={assistant_id}")
|
||||
assistant_info = AssistantDao.get_one_assistant(assistant_id)
|
||||
cls._build_log(user, ip_address, EventType.CREATE_BUILD, ObjectType.ASSISTANT,
|
||||
assistant_info.id, assistant_info.name, ResourceTypeEnum.ASSISTANT)
|
||||
|
||||
@classmethod
|
||||
def update_build_assistant(cls, user: UserPayload, ip_address: str, assistant_id: str):
|
||||
"""
|
||||
更新助手的审计日志
|
||||
"""
|
||||
logger.info(f"act=update_build_assistant user={user.user_name} ip={ip_address} assistant={assistant_id}")
|
||||
assistant_info = AssistantDao.get_one_assistant(assistant_id)
|
||||
|
||||
cls._build_log(user, ip_address, EventType.UPDATE_BUILD, ObjectType.ASSISTANT,
|
||||
assistant_info.id, assistant_info.name, ResourceTypeEnum.ASSISTANT)
|
||||
|
||||
@classmethod
|
||||
def delete_build_assistant(cls, user: UserPayload, ip_address: str, assistant_id: str):
|
||||
"""
|
||||
删除助手的审计日志
|
||||
"""
|
||||
logger.info(f"act=delete_build_assistant user={user.user_name} ip={ip_address} assistant={assistant_id}")
|
||||
assistant_info = AssistantDao.get_one_assistant(assistant_id)
|
||||
|
||||
cls._build_log(user, ip_address, EventType.DELETE_BUILD, ObjectType.ASSISTANT,
|
||||
assistant_info.id, assistant_info.name, ResourceTypeEnum.ASSISTANT)
|
||||
|
||||
@classmethod
|
||||
def _knowledge_log(cls, user: UserPayload, ip_address: str, event_type: EventType, object_type: ObjectType,
|
||||
object_id: str, object_name: str, resource_type: ResourceTypeEnum, resource_id: str):
|
||||
"""
|
||||
知识库模块的日志
|
||||
"""
|
||||
# 获取资源属于哪些用户组
|
||||
groups = GroupResourceDao.get_resource_group(resource_type, resource_id)
|
||||
group_ids = [one.group_id for one in groups]
|
||||
|
||||
# 插入审计日志
|
||||
audit_log = AuditLog(
|
||||
operator_id=user.user_id,
|
||||
operator_name=user.user_name,
|
||||
group_ids=group_ids,
|
||||
system_id=SystemId.KNOWLEDGE.value,
|
||||
event_type=event_type.value,
|
||||
object_type=object_type.value,
|
||||
object_id=object_id,
|
||||
object_name=object_name,
|
||||
ip_address=ip_address,
|
||||
)
|
||||
AuditLogDao.insert_audit_logs([audit_log])
|
||||
|
||||
@classmethod
|
||||
def create_knowledge(cls, user: UserPayload, ip_address: str, knowledge_id: int):
|
||||
"""
|
||||
新建知识库的审计日志
|
||||
"""
|
||||
logger.info(f"act=create_knowledge user={user.user_name} ip={ip_address} knowledge={knowledge_id}")
|
||||
knowledge_info = KnowledgeDao.query_by_id(knowledge_id)
|
||||
cls._knowledge_log(user, ip_address, EventType.CREATE_KNOWLEDGE, ObjectType.KNOWLEDGE,
|
||||
str(knowledge_id), knowledge_info.name, ResourceTypeEnum.KNOWLEDGE, str(knowledge_id))
|
||||
|
||||
@classmethod
|
||||
def delete_knowledge(cls, user: UserPayload, ip_address: str, knowledge: Knowledge):
|
||||
"""
|
||||
删除知识库的审计日志
|
||||
"""
|
||||
logger.info(f"act=delete_knowledge user={user.user_name} ip={ip_address} knowledge={knowledge.id}")
|
||||
cls._knowledge_log(user, ip_address, EventType.DELETE_KNOWLEDGE, ObjectType.KNOWLEDGE,
|
||||
str(knowledge.id), knowledge.name, ResourceTypeEnum.KNOWLEDGE, str(knowledge.id))
|
||||
|
||||
@classmethod
|
||||
def upload_knowledge_file(cls, user: UserPayload, ip_address: str, knowledge_id: int, file_name: str):
|
||||
"""
|
||||
知识库上传文件的审计日志
|
||||
"""
|
||||
logger.info(f"act=upload_knowledge_file user={user.user_name} ip={ip_address}"
|
||||
f" knowledge={knowledge_id} file={file_name}")
|
||||
cls._knowledge_log(user, ip_address, EventType.UPLOAD_FILE, ObjectType.FILE,
|
||||
str(knowledge_id), file_name, ResourceTypeEnum.KNOWLEDGE, str(knowledge_id))
|
||||
|
||||
@classmethod
|
||||
def delete_knowledge_file(cls, user: UserPayload, ip_address: str, knowledge_id: int, file_name: str):
|
||||
"""
|
||||
知识库删除文件的审计日志
|
||||
"""
|
||||
logger.info(f"act=delete_knowledge_file user={user.user_name} ip={ip_address}"
|
||||
f" knowledge={knowledge_id} file={file_name}")
|
||||
cls._knowledge_log(user, ip_address, EventType.DELETE_FILE, ObjectType.FILE,
|
||||
str(knowledge_id), file_name, ResourceTypeEnum.KNOWLEDGE, str(knowledge_id))
|
||||
|
||||
@classmethod
|
||||
def _system_log(cls, user: UserPayload, ip_address: str, group_ids: List[int], event_type: EventType,
|
||||
object_type: ObjectType, object_id: str, object_name: str, note: str = ''):
|
||||
|
||||
audit_log = AuditLog(
|
||||
operator_id=user.user_id,
|
||||
operator_name=user.user_name,
|
||||
group_ids=group_ids,
|
||||
system_id=SystemId.SYSTEM.value,
|
||||
event_type=event_type.value,
|
||||
object_type=object_type.value,
|
||||
object_id=object_id,
|
||||
object_name=object_name,
|
||||
ip_address=ip_address,
|
||||
note=note,
|
||||
)
|
||||
AuditLogDao.insert_audit_logs([audit_log])
|
||||
|
||||
@classmethod
|
||||
def update_user(cls, user: UserPayload, ip_address: str, user_id: int, group_ids: List[int], note: str):
|
||||
"""
|
||||
修改用户的用户组和角色
|
||||
"""
|
||||
logger.info(f"act=update_system_user user={user.user_name} ip={ip_address} user_id={user_id} note={note}")
|
||||
user_info = UserDao.get_user(user_id)
|
||||
cls._system_log(user, ip_address, group_ids, EventType.UPDATE_USER,
|
||||
ObjectType.USER_CONF, str(user_id), user_info.user_name, note)
|
||||
|
||||
@classmethod
|
||||
def forbid_user(cls, user: UserPayload, ip_address: str, user_info: User):
|
||||
"""
|
||||
user: 操作用户
|
||||
user_info: 被操作用户
|
||||
"""
|
||||
logger.info(f"act=forbid_user user={user.user_name} ip={ip_address} user_id={user.user_id}")
|
||||
# 获取用户所属的分组
|
||||
user_group = UserGroupDao.get_user_group(user_info.user_id)
|
||||
user_group = [one.group_id for one in user_group]
|
||||
cls._system_log(user, ip_address, user_group, EventType.FORBID_USER,
|
||||
ObjectType.USER_CONF, str(user_info.user_id), user_info.user_name)
|
||||
|
||||
@classmethod
|
||||
def recover_user(cls, user: UserPayload, ip_address: str, user_info: User):
|
||||
logger.info(f"act=recover_user user={user.user_name} ip={ip_address} user_id={user_info.user_id}")
|
||||
# 获取用户所属的分组
|
||||
user_group = UserGroupDao.get_user_group(user_info.user_id)
|
||||
user_group = [one.group_id for one in user_group]
|
||||
cls._system_log(user, ip_address, user_group, EventType.RECOVER_USER,
|
||||
ObjectType.USER_CONF, str(user_info.user_id), user_info.user_name)
|
||||
|
||||
@classmethod
|
||||
def create_user_group(cls, user: UserPayload, ip_address: str, group_info: Group):
|
||||
logger.info(f"act=create_user_group user={user.user_name} ip={ip_address} group_id={group_info.id}")
|
||||
cls._system_log(user, ip_address, [group_info.id], EventType.CREATE_USER_GROUP,
|
||||
ObjectType.USER_GROUP_CONF, str(group_info.id), group_info.group_name)
|
||||
|
||||
@classmethod
|
||||
def update_user_group(cls, user: UserPayload, ip_address: str, group_info: Group):
|
||||
logger.info(f"act=update_user_group user={user.user_name} ip={ip_address} group_id={group_info.id}")
|
||||
# 获取用户组信息
|
||||
cls._system_log(user, ip_address, [group_info.id], EventType.UPDATE_USER_GROUP,
|
||||
ObjectType.USER_GROUP_CONF, str(group_info.id), group_info.group_name)
|
||||
|
||||
@classmethod
|
||||
def delete_user_group(cls, user: UserPayload, ip_address: str, group_info: Group):
|
||||
logger.info(f"act=delete_user_group user={user.user_name} ip={ip_address} group_id={group_info.id}")
|
||||
# 获取用户组信息
|
||||
cls._system_log(user, ip_address, [group_info.id], EventType.DELETE_USER_GROUP,
|
||||
ObjectType.USER_GROUP_CONF, str(group_info.id), group_info.group_name)
|
||||
|
||||
@classmethod
|
||||
def create_role(cls, user: UserPayload, ip_address: str, role: Role):
|
||||
logger.info(f"act=create_role user={user.user_name} ip={ip_address} role_id={role.id}")
|
||||
|
||||
cls._system_log(user, ip_address, [role.group_id], EventType.CREATE_ROLE,
|
||||
ObjectType.ROLE_CONF, str(role.id), role.role_name)
|
||||
|
||||
@classmethod
|
||||
def update_role(cls, user: UserPayload, ip_address: str, role: Role):
|
||||
logger.info(f"act=update_role user={user.user_name} ip={ip_address} role_id={role.id}")
|
||||
|
||||
cls._system_log(user, ip_address, [role.group_id], EventType.UPDATE_ROLE,
|
||||
ObjectType.ROLE_CONF, str(role.id), role.role_name)
|
||||
|
||||
@classmethod
|
||||
def delete_role(cls, user: UserPayload, ip_address: str, role: Role):
|
||||
logger.info(f"act=delete_role user={user.user_name} ip={ip_address} role_id={role.id}")
|
||||
|
||||
cls._system_log(user, ip_address, [role.group_id], EventType.DELETE_ROLE,
|
||||
ObjectType.ROLE_CONF, str(role.id), role.role_name)
|
||||
|
||||
@classmethod
|
||||
def user_login(cls, user: UserPayload, ip_address: str):
|
||||
logger.info(f"act=user_login user={user.user_name} ip={ip_address} user_id={user.user_id}")
|
||||
# 获取用户所属的分组
|
||||
user_group = UserGroupDao.get_user_group(user.user_id)
|
||||
user_group = [one.group_id for one in user_group]
|
||||
cls._system_log(user, ip_address, user_group, EventType.USER_LOGIN,
|
||||
ObjectType.NONE, '', '')
|
||||
|
||||
@classmethod
|
||||
def get_filter_flow_ids(cls, user: UserPayload, flow_ids: List[str], group_ids: List[int]) -> (bool, List):
|
||||
""" 通过flow_ids和group_ids获取最终的 技能id筛选条件 false: 表示返回空列表"""
|
||||
flow_ids = [one for one in flow_ids]
|
||||
group_admins = []
|
||||
if not user.is_admin():
|
||||
user_groups = UserGroupDao.get_user_admin_group(user.user_id)
|
||||
# 不是用户组管理员,没有权限
|
||||
if not user_groups:
|
||||
raise UnAuthorizedError.http_exception()
|
||||
group_admins = [one.group_id for one in user_groups]
|
||||
# 分组id做交集
|
||||
if group_ids:
|
||||
if group_admins:
|
||||
# 查询了不属于用户管理的用户组,返回为空
|
||||
group_admins = list(set(group_admins) & set(group_ids))
|
||||
if len(group_admins) == 0:
|
||||
return False, []
|
||||
else:
|
||||
group_admins = group_ids
|
||||
|
||||
# 获取分组下所有的应用ID
|
||||
group_flows = []
|
||||
if group_admins:
|
||||
group_flows = GroupResourceDao.get_groups_resource(group_admins,
|
||||
resource_types=[ResourceTypeEnum.FLOW,
|
||||
ResourceTypeEnum.WORK_FLOW,
|
||||
ResourceTypeEnum.ASSISTANT])
|
||||
# 用户管理下的用户组没有资源
|
||||
if not group_flows:
|
||||
return False, []
|
||||
group_flows = [one.third_id for one in group_flows]
|
||||
|
||||
# 获取最终的技能ID限制列表
|
||||
filter_flow_ids = []
|
||||
if flow_ids and group_flows:
|
||||
filter_flow_ids = list(set(group_flows) & set(flow_ids))
|
||||
if not filter_flow_ids:
|
||||
return False, []
|
||||
elif flow_ids:
|
||||
filter_flow_ids = flow_ids
|
||||
elif group_flows:
|
||||
filter_flow_ids = group_flows
|
||||
return True, filter_flow_ids
|
||||
|
||||
@classmethod
|
||||
def get_session_list(cls, user: UserPayload, flow_ids: List[str], user_ids: List[int], group_ids: List[int],
|
||||
start_date: datetime, end_date: datetime,
|
||||
feedback: str, sensitive_status: int, page: int, page_size: int) -> (list, int):
|
||||
flag, filter_flow_ids = cls.get_filter_flow_ids(user, flow_ids, group_ids)
|
||||
if not flag:
|
||||
return [], 0
|
||||
filter_status = []
|
||||
if sensitive_status:
|
||||
filter_status = [SensitiveStatus(sensitive_status)]
|
||||
|
||||
res = MessageSessionDao.filter_session(sensitive_status=filter_status, feedback=feedback,
|
||||
flow_ids=filter_flow_ids, user_ids=user_ids, start_date=start_date,
|
||||
end_date=end_date, page=page, limit=page_size)
|
||||
total = MessageSessionDao.filter_session_count(sensitive_status=filter_status, feedback=feedback,
|
||||
flow_ids=filter_flow_ids, user_ids=user_ids,
|
||||
start_date=start_date,
|
||||
end_date=end_date)
|
||||
|
||||
res_users = []
|
||||
for one in res:
|
||||
res_users.append(one.user_id)
|
||||
user_list = UserDao.get_user_by_ids(res_users)
|
||||
user_map = {user.user_id: user.user_name for user in user_list}
|
||||
result = []
|
||||
for one in res:
|
||||
result.append(AppChatList(**one.model_dump(),
|
||||
like_count=one.like,
|
||||
dislike_count=one.dislike,
|
||||
copied_count=one.copied,
|
||||
user_name=user_map.get(one.user_id, one.user_id),
|
||||
user_groups=user.get_user_groups(one.user_id)))
|
||||
|
||||
return result, total
|
||||
|
||||
@classmethod
|
||||
def get_session_messages(cls, user: UserPayload, flow_ids: List[str], user_ids: List[int], group_ids: List[int],
|
||||
start_date: datetime, end_date: datetime, feedback: str,
|
||||
sensitive_status: int) -> List[AppChatList]:
|
||||
page = 1
|
||||
page_size = 50
|
||||
res = []
|
||||
while True:
|
||||
result, total = cls.get_session_list(user, flow_ids, user_ids, group_ids, start_date, end_date, feedback,
|
||||
sensitive_status, page, page_size)
|
||||
if not result:
|
||||
break
|
||||
page += 1
|
||||
res.extend(cls.get_chat_messages(result))
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
def export_session_messages(cls, user: UserPayload, flow_ids: List[str], user_ids: List[int],
|
||||
group_ids: List[int],
|
||||
start_date: datetime, end_date: datetime,
|
||||
feedback: str, sensitive_status: int) -> str:
|
||||
page = 1
|
||||
page_size = 30
|
||||
excel_data = [
|
||||
['会话ID', '应用名称', '会话创建时间', '用户名称', '消息角色', '消息发送时间', '消息文本内容', '点赞',
|
||||
'点踩', '复制']]
|
||||
bisheng_pro = settings.get_system_login_method().bisheng_pro
|
||||
if bisheng_pro:
|
||||
excel_data[0].append('是否命中内容安全审查')
|
||||
|
||||
while True:
|
||||
result, total = cls.get_session_list(user, flow_ids, user_ids, group_ids, start_date, end_date, feedback,
|
||||
sensitive_status, page, page_size)
|
||||
if not result:
|
||||
break
|
||||
page += 1
|
||||
chat_list = cls.get_chat_messages(result)
|
||||
for chat in chat_list:
|
||||
for message in chat.messages:
|
||||
message_data = [chat.chat_id, chat.flow_name, chat.create_time.strftime('%Y/%m/%d %H:%M:%S'),
|
||||
chat.user_name,
|
||||
'用户' if message.category == 'question' else 'AI',
|
||||
message.create_time.strftime('%Y/%m/%d %H:%M:%S'),
|
||||
message.message,
|
||||
'是' if message.liked == LikedType.LIKED.value else '否',
|
||||
'是' if message.liked == LikedType.DISLIKED.value else '否',
|
||||
'是' if message.copied else '否']
|
||||
if bisheng_pro:
|
||||
message_data.append(
|
||||
'是' if message.sensitive_status == SensitiveStatus.VIOLATIONS.value else '否')
|
||||
excel_data.append(message_data)
|
||||
|
||||
minio_client = MinioClient()
|
||||
tmp_object_name = f'tmp/session/export_{generate_uuid()}.csv'
|
||||
with NamedTemporaryFile(mode='w', newline='') as tmp_file:
|
||||
csv_writer = csv.writer(tmp_file)
|
||||
csv_writer.writerows(excel_data)
|
||||
tmp_file.seek(0)
|
||||
minio_client.upload_minio(tmp_object_name, tmp_file.name,
|
||||
'application/text',
|
||||
minio_client.tmp_bucket)
|
||||
share_url = minio_client.get_share_link(tmp_object_name, minio_client.tmp_bucket)
|
||||
return minio_client.clear_minio_share_host(share_url)
|
||||
|
||||
@classmethod
|
||||
def get_chat_messages(cls, chat_list: List[AppChatList]) -> List[AppChatList]:
|
||||
chat_ids = [chat.chat_id for chat in chat_list]
|
||||
|
||||
chat_messages = ChatMessageDao.get_all_message_by_chat_ids(chat_ids)
|
||||
chat_messages_map = {}
|
||||
for one in chat_messages:
|
||||
if one.chat_id not in chat_messages_map:
|
||||
chat_messages_map[one.chat_id] = []
|
||||
chat_messages_map[one.chat_id].append(one)
|
||||
for chat in chat_list:
|
||||
chat_messages = chat_messages_map.get(chat.chat_id, [])
|
||||
# remove workflow input event, because it's not show in web
|
||||
chat.messages = [message for message in chat_messages
|
||||
if message.category != WorkflowEventType.UserInput.value]
|
||||
return chat_list
|
||||
@@ -0,0 +1,34 @@
|
||||
from bisheng.cache import InMemoryCache
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.settings import settings
|
||||
from bisheng.utils.minio_client import MinioClient
|
||||
|
||||
|
||||
class BaseService:
|
||||
LogoMemoryCache = InMemoryCache(max_size=200, expiration_time=3600 * 24)
|
||||
|
||||
@classmethod
|
||||
def get_logo_share_link(cls, logo_path: str):
|
||||
if not logo_path:
|
||||
return ''
|
||||
cache_key = f'logo_cache:{logo_path}'
|
||||
# 先从内存中获取
|
||||
share_url = cls.LogoMemoryCache.get(cache_key)
|
||||
if share_url:
|
||||
return share_url
|
||||
|
||||
# 再从redis缓存中获取
|
||||
share_url = redis_client.get(cache_key)
|
||||
if share_url:
|
||||
cls.LogoMemoryCache.set(cache_key, share_url)
|
||||
return share_url
|
||||
|
||||
minio_client = MinioClient()
|
||||
share_url = minio_client.get_share_link(logo_path)
|
||||
# 去除前缀通过nginx访问,防止访问不到文件
|
||||
share_url = minio_client.clear_minio_share_host(share_url)
|
||||
|
||||
# 缓存5天, 临时链接有效期为7天
|
||||
redis_client.set(cache_key, share_url, 3600 * 120)
|
||||
cls.LogoMemoryCache.set(cache_key, share_url)
|
||||
return share_url
|
||||
@@ -1,5 +1,82 @@
|
||||
import asyncio
|
||||
import json
|
||||
# 设置 websockets 的日志级别为 NONE
|
||||
import logging
|
||||
from collections import defaultdict
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
from bisheng.api.v1.schemas import resp_500
|
||||
from bisheng.database.base import session_getter
|
||||
from bisheng.database.models.message import ChatMessage
|
||||
from pydantic import BaseModel
|
||||
from websockets import connect
|
||||
|
||||
# 维护一个连接池
|
||||
connection_pool = defaultdict(asyncio.Queue)
|
||||
logging.getLogger('websockets').setLevel(logging.ERROR)
|
||||
|
||||
expire = 600 # reids 60s 过期
|
||||
|
||||
|
||||
class TimedQueue:
|
||||
|
||||
def __init__(self):
|
||||
self.queue = asyncio.Queue()
|
||||
self.last_active = datetime.now()
|
||||
|
||||
async def put_nowait(self, item):
|
||||
self.last_active = datetime.now()
|
||||
await self.queue.put(item)
|
||||
|
||||
async def get_nowait(self):
|
||||
self.last_active = datetime.now()
|
||||
return await self.queue.get()
|
||||
|
||||
def empty(self):
|
||||
return self.queue.empty()
|
||||
|
||||
def qsize(self):
|
||||
return self.queue.qsize()
|
||||
|
||||
|
||||
async def clean_inactive_queues(queue: defaultdict, timeout_threshold: timedelta):
|
||||
while True:
|
||||
current_time = datetime.now()
|
||||
for key, timed_queue in list(queue.items()):
|
||||
# 如果队列超过设定的阈值时间没有活跃,则清理队列
|
||||
if current_time - timed_queue.last_active > timeout_threshold:
|
||||
while not timed_queue.empty():
|
||||
timed_queue.get_nowait() # 从队列中移除任务
|
||||
del queue[key] # 删除队列
|
||||
await asyncio.sleep(timeout_threshold.total_seconds())
|
||||
|
||||
|
||||
# 维护一个连接池
|
||||
connection_pool = defaultdict(TimedQueue)
|
||||
# clean_inactive_queues(connection_pool, timedelta(minutes=5))
|
||||
|
||||
|
||||
async def get_connection(uri, identifier):
|
||||
"""
|
||||
获取WebSocket连接。如果连接池中有可用的连接,则直接返回;
|
||||
否则,创建新的连接并添加到连接池。
|
||||
"""
|
||||
if connection_pool[identifier].empty():
|
||||
# 建立新的WebSocket连接
|
||||
websocket = await connect(uri)
|
||||
|
||||
await connection_pool[identifier].put_nowait(websocket)
|
||||
|
||||
# 从连接池中获取连接
|
||||
websocket = await connection_pool[identifier].get_nowait()
|
||||
return websocket
|
||||
|
||||
|
||||
async def release_connection(identifier, websocket):
|
||||
"""
|
||||
释放WebSocket连接,将其放回连接池。
|
||||
"""
|
||||
await connection_pool[identifier].put_nowait(websocket)
|
||||
|
||||
|
||||
def comment_answer(message_id: int, comment: str):
|
||||
@@ -9,3 +86,68 @@ def comment_answer(message_id: int, comment: str):
|
||||
message.remark = comment[:4096]
|
||||
session.add(message)
|
||||
session.commit()
|
||||
|
||||
|
||||
class ContentStreamResp(BaseModel):
|
||||
role: str
|
||||
content: str
|
||||
|
||||
|
||||
class ChoiceStreamResp(BaseModel):
|
||||
index: int = 0
|
||||
delta: ContentStreamResp = 0
|
||||
session_id: str
|
||||
|
||||
def __str__(self) -> str:
|
||||
jsonData = '{"index": "%s", "delta": %s, "session_id": "%s"}' % (
|
||||
self.index, json.dumps(self.delta.dict(), ensure_ascii=False), self.session_id)
|
||||
return '{"choices":[%s]}\n\n' % (jsonData)
|
||||
|
||||
|
||||
async def event_stream(
|
||||
webosocket: connect,
|
||||
message: str,
|
||||
session_id: str,
|
||||
model: str,
|
||||
streaming: bool,
|
||||
):
|
||||
|
||||
payload = {'inputs': message, 'flow_id': model, 'chat_id': session_id}
|
||||
try:
|
||||
await webosocket.send(json.dumps(payload, ensure_ascii=False))
|
||||
except Exception as e:
|
||||
yield json.dumps(resp_500(message=str(e)).__dict__)
|
||||
return
|
||||
sync = ''
|
||||
while True:
|
||||
try:
|
||||
msg = await webosocket.recv()
|
||||
except Exception as e:
|
||||
yield json.dumps(resp_500(message=str(e)).__dict__)
|
||||
break
|
||||
if msg is None:
|
||||
continue
|
||||
# 判断msg 的类型
|
||||
res = json.loads(msg)
|
||||
if streaming:
|
||||
if res.get('type') != 'end' and res.get('message'):
|
||||
delta = ContentStreamResp(role='assistant', content=res.get('message'))
|
||||
yield str(ChoiceStreamResp(index=0, session_id=session_id, delta=delta))
|
||||
else:
|
||||
# 通过此处控制下面的close是否发送消息
|
||||
if res.get('type') == 'end':
|
||||
sync = res.get('message')
|
||||
|
||||
if res.get('type') == 'close':
|
||||
if not streaming and sync:
|
||||
delta = ContentStreamResp(role='assistant', content=sync)
|
||||
msg = ChoiceStreamResp(index=0,
|
||||
session_id=session_id,
|
||||
delta=delta,
|
||||
finish_reason='stop')
|
||||
yield '{"choices":[%s]}' % (json.dumps(msg.dict()))
|
||||
# 释放连接
|
||||
elif streaming:
|
||||
yield 'data: [DONE]'
|
||||
await release_connection(session_id, webosocket)
|
||||
break
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from bisheng.api.services.base import BaseService
|
||||
from bisheng.api.v1.schema.dataset_param import CreateDatasetParam
|
||||
from bisheng.database.models.dataset import Dataset, DatasetCreate, DatasetDao, DatasetRead
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.utils.minio_client import MinioClient
|
||||
from fastapi import HTTPException
|
||||
|
||||
|
||||
class DatasetService(BaseService):
|
||||
|
||||
@classmethod
|
||||
def build_dataset_list(cls,
|
||||
page: int,
|
||||
limit: int,
|
||||
keyword: Optional[str] = None) -> List[Dict]:
|
||||
"""补全list 数据"""
|
||||
|
||||
dataset_list = DatasetDao.filter_dataset_by_ids(dataset_ids=[],
|
||||
keyword=keyword,
|
||||
page=page,
|
||||
limit=limit)
|
||||
count_filter = []
|
||||
if keyword:
|
||||
count_filter.append(Dataset.name.like('%{}%'.format(keyword)))
|
||||
total_count = DatasetDao.get_count_by_filter(count_filter)
|
||||
|
||||
user_ids = [one.user_id for one in dataset_list]
|
||||
user_list = UserDao.get_user_by_ids(user_ids)
|
||||
user_dict = {one.user_id: one for one in user_list}
|
||||
res = [DatasetRead.validate(one) for one in dataset_list]
|
||||
for one in res:
|
||||
one.user_name = user_dict[one.user_id].user_name
|
||||
if one.object_name:
|
||||
one.url = MinioClient().get_share_link(one.object_name)
|
||||
|
||||
return res, total_count
|
||||
|
||||
@classmethod
|
||||
def create_dataset(cls, user_id: int, data: CreateDatasetParam):
|
||||
"""创建数据集"""
|
||||
dataset_insert = DatasetCreate.validate(data)
|
||||
dataset_insert.user_id = user_id
|
||||
isExist = DatasetDao.get_dataset_by_name(data.name)
|
||||
if isExist:
|
||||
raise ValueError('数据集名称已存在')
|
||||
dataset = DatasetDao.insert(dataset_insert)
|
||||
# 处理文件
|
||||
object_name = f'/dataset/{dataset.id}/{dataset.name}'
|
||||
if data.file_url:
|
||||
|
||||
# MinioClient().upload_minio()
|
||||
dataset.object_name = object_name
|
||||
if data.qa_list:
|
||||
for qa in data.qa_list:
|
||||
qa.dataset_id = dataset.id
|
||||
# QADao.insert(qa)
|
||||
|
||||
dataset = DatasetDao.update(dataset)
|
||||
return dataset
|
||||
|
||||
@classmethod
|
||||
def delete_dataset(cls, dataset_id: int):
|
||||
dataset = DatasetDao.get_dataset_by_id(dataset_id)
|
||||
if not dataset:
|
||||
raise HTTPException(status_code=404, detail='Dataset not found')
|
||||
# 处理minio
|
||||
object_name = dataset.object_name
|
||||
if object_name:
|
||||
minio_client = MinioClient()
|
||||
minio_client.delete_minio(object_name)
|
||||
DatasetDao.delete(dataset)
|
||||
return True
|
||||
@@ -0,0 +1,340 @@
|
||||
# flake8: noqa
|
||||
"""Loads PDF with semantic splilter."""
|
||||
import base64
|
||||
import logging
|
||||
import os
|
||||
from typing import List
|
||||
from uuid import uuid4
|
||||
|
||||
import cv2
|
||||
import fitz
|
||||
import requests
|
||||
from PIL import Image
|
||||
from langchain_community.docstore.document import Document
|
||||
from langchain_community.document_loaders.pdf import BasePDFLoader
|
||||
|
||||
from bisheng.utils.minio_client import minio_client
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def get_image_tag(results, part):
|
||||
element_id = part.get("element_id", None)
|
||||
url = results.get(element_id)
|
||||
return f""
|
||||
|
||||
|
||||
def get_image_parts(partitions):
|
||||
page_dict = {}
|
||||
for part in partitions:
|
||||
label = part["type"]
|
||||
if label == "Image":
|
||||
bboxes = part.get("metadata", {}).get("extra_data", {}).get("bboxes", [])
|
||||
page = part.get("metadata", {}).get("extra_data", {}).get("pages", -1)
|
||||
element_id = part.get("element_id", None)
|
||||
if len(bboxes) == 0 or page == -1 or not element_id:
|
||||
continue
|
||||
item = {}
|
||||
item["bboxes"] = bboxes[0]
|
||||
item["element_id"] = element_id
|
||||
page_id = page[0]
|
||||
if page_id not in page_dict:
|
||||
page_dict[page_id] = []
|
||||
page_dict[page_id].append(item)
|
||||
return page_dict
|
||||
|
||||
|
||||
def crop_image(image_file, item, cropped_imag_base_dir):
|
||||
element_id = item.get("element_id")
|
||||
bbox = item.get("bboxes")
|
||||
img = cv2.imread(image_file)
|
||||
x1, y1, x2, y2 = bbox
|
||||
cropped_img = img[y1:y2, x1:x2]
|
||||
file_name = f"{element_id}.png"
|
||||
cv2.imwrite(os.path.join(cropped_imag_base_dir, file_name), cropped_img)
|
||||
return file_name
|
||||
|
||||
|
||||
def extract_pdf_images(file_name, page_dict, doc_id, knowledge_id):
|
||||
from bisheng.api.services.knowledge_imp import put_images_to_minio
|
||||
from bisheng.api.services.knowledge_imp import KnowledgeUtils
|
||||
from bisheng.cache.utils import CACHE_DIR
|
||||
|
||||
result = {}
|
||||
base_dir = f"{CACHE_DIR}/{doc_id}"
|
||||
cropped_image_base_dir = f"{base_dir}/images"
|
||||
pdf_page_base_dir = f"{base_dir}/images"
|
||||
|
||||
if not os.path.exists(pdf_page_base_dir):
|
||||
os.makedirs(pdf_page_base_dir)
|
||||
if not os.path.exists(cropped_image_base_dir):
|
||||
os.makedirs(cropped_image_base_dir)
|
||||
|
||||
pdf_document = fitz.open(file_name)
|
||||
for page_number, items in page_dict.items():
|
||||
page = pdf_document[page_number]
|
||||
pix = page.get_pixmap()
|
||||
image = Image.frombytes("RGB", (pix.width, pix.height), pix.samples)
|
||||
pdf_image_file_name = f"{pdf_page_base_dir}/{page_number}.png"
|
||||
image.save(pdf_image_file_name)
|
||||
for item in items:
|
||||
cropped_image_file = crop_image(
|
||||
pdf_image_file_name, item, cropped_image_base_dir
|
||||
)
|
||||
result[item["element_id"]] = (
|
||||
f"/{minio_client.bucket}/{KnowledgeUtils.get_knowledge_file_image_dir(doc_id, knowledge_id)}/{cropped_image_file}"
|
||||
)
|
||||
put_images_to_minio(cropped_image_base_dir, knowledge_id, doc_id)
|
||||
return result
|
||||
|
||||
|
||||
def pre_handle(partitions, file_name, knowledge_id):
|
||||
doc_id = str(uuid4())
|
||||
image_parts = get_image_parts(partitions=partitions)
|
||||
if len(image_parts) == 0:
|
||||
return []
|
||||
return extract_pdf_images(file_name, image_parts, doc_id, knowledge_id)
|
||||
|
||||
|
||||
def merge_partitions(file_name, partitions, knowledge_id=None):
|
||||
# 预处理pdf,提取图片
|
||||
pre_handle_results = pre_handle(
|
||||
partitions=partitions, file_name=file_name, knowledge_id=knowledge_id
|
||||
)
|
||||
text_elem_sep = "\n"
|
||||
doc_content = []
|
||||
is_first_elem = True
|
||||
last_label = ""
|
||||
prev_length = 0
|
||||
metadata = dict(bboxes=[], pages=[], indexes=[], types=[])
|
||||
|
||||
for part in partitions:
|
||||
label, text = part["type"], part["text"]
|
||||
extra_data = part["metadata"]["extra_data"]
|
||||
if label == "Image":
|
||||
part["text"] = get_image_tag(pre_handle_results, part)
|
||||
text = part["text"]
|
||||
|
||||
if is_first_elem:
|
||||
f_text = text + "\n" if label == "Title" else text
|
||||
doc_content.append(f_text)
|
||||
is_first_elem = False
|
||||
else:
|
||||
if last_label == "Title" and label == "Title":
|
||||
doc_content.append("\n" + text)
|
||||
elif label == "Title":
|
||||
doc_content.append("\n\n" + text)
|
||||
elif label == "Table":
|
||||
doc_content.append("\n\n" + text)
|
||||
else:
|
||||
if last_label == "Table":
|
||||
doc_content.append(text_elem_sep * 2 + text)
|
||||
else:
|
||||
doc_content.append(text_elem_sep + text)
|
||||
|
||||
last_label = label
|
||||
metadata["bboxes"].extend(
|
||||
list(map(lambda x: list(map(int, x)), extra_data["bboxes"]))
|
||||
)
|
||||
metadata["pages"].extend(extra_data["pages"])
|
||||
metadata["types"].extend(extra_data["types"])
|
||||
|
||||
indexes = extra_data["indexes"]
|
||||
up_indexes = [[s + prev_length, e + prev_length] for (s, e) in indexes]
|
||||
metadata["indexes"].extend(up_indexes)
|
||||
prev_length += len(doc_content[-1])
|
||||
|
||||
content = "".join(doc_content)
|
||||
return content, metadata
|
||||
|
||||
|
||||
class Etl4lmLoader(BasePDFLoader):
|
||||
"""Loads a PDF with pypdf and chunks at character level. dummy version
|
||||
|
||||
Loader also stores page numbers in metadata.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
file_name: str,
|
||||
file_path: str,
|
||||
unstructured_api_key: str = None,
|
||||
unstructured_api_url: str = None,
|
||||
force_ocr: bool = False,
|
||||
enable_formular: bool = True,
|
||||
filter_page_header_footer: bool = False,
|
||||
ocr_sdk_url: str = None,
|
||||
timeout: int = 60,
|
||||
knowledge_id: int = None,
|
||||
start: int = 0,
|
||||
n: int = None,
|
||||
verbose: bool = False,
|
||||
kwargs: dict = {},
|
||||
) -> None:
|
||||
"""Initialize with a file path."""
|
||||
self.unstructured_api_url = unstructured_api_url
|
||||
self.unstructured_api_key = unstructured_api_key
|
||||
self.force_ocr = force_ocr
|
||||
self.enable_formular = enable_formular
|
||||
self.filter_page_header_footer = filter_page_header_footer
|
||||
self.ocr_sdk_url = ocr_sdk_url
|
||||
self.headers = {"Content-Type": "application/json"}
|
||||
self.file_name = file_name
|
||||
self.timemout = timeout
|
||||
self.start = start
|
||||
self.n = n
|
||||
self.extra_kwargs = kwargs
|
||||
self.partitions = None
|
||||
self.knowledge_id = knowledge_id
|
||||
super().__init__(file_path)
|
||||
|
||||
def load(self) -> List[Document]:
|
||||
"""Load given path as pages."""
|
||||
b64_data = base64.b64encode(open(self.file_path, "rb").read()).decode()
|
||||
parameters = {"start": self.start, "n": self.n}
|
||||
parameters.update(self.extra_kwargs)
|
||||
# TODO: add filter_page_header_footer into payload when elt4llm is ready.
|
||||
payload = dict(
|
||||
filename=os.path.basename(self.file_name),
|
||||
b64_data=[b64_data],
|
||||
mode="partition",
|
||||
force_ocr=self.force_ocr,
|
||||
enable_formula=self.enable_formular,
|
||||
ocr_sdk_url=self.ocr_sdk_url,
|
||||
parameters=parameters,
|
||||
)
|
||||
try:
|
||||
resp = requests.post(
|
||||
self.unstructured_api_url, headers=self.headers, json=payload, timeout=self.timemout
|
||||
)
|
||||
except requests.Timeout as e:
|
||||
logger.error(f"Request to etl4lm API timed out: {e}")
|
||||
raise Exception("etl4lm服务繁忙,请升级etl4lm服务的算力")
|
||||
if resp.status_code != 200:
|
||||
raise Exception(
|
||||
f"file partition {os.path.basename(self.file_name)} failed resp={resp.text}"
|
||||
)
|
||||
|
||||
resp = resp.json()
|
||||
if 200 != resp.get("status_code"):
|
||||
logger.info(
|
||||
f"file partition {os.path.basename(self.file_name)} error resp={resp}"
|
||||
)
|
||||
raise Exception(
|
||||
f"file partition error {os.path.basename(self.file_name)} error resp={resp}"
|
||||
)
|
||||
partitions = resp["partitions"]
|
||||
if partitions:
|
||||
logger.info(f"content_from_partitions")
|
||||
self.partitions = partitions
|
||||
content, metadata = merge_partitions(
|
||||
self.file_path, partitions, self.knowledge_id
|
||||
)
|
||||
elif resp.get("text"):
|
||||
logger.info(f"content_from_text")
|
||||
content = resp["text"]
|
||||
metadata = {
|
||||
"bboxes": [],
|
||||
"pages": [],
|
||||
"indexes": [],
|
||||
"types": [],
|
||||
}
|
||||
else:
|
||||
logger.warning(f"content_is_empty resp={resp}")
|
||||
content = ""
|
||||
metadata = {}
|
||||
|
||||
logger.info(f'unstruct_return code={resp.get("status_code")}')
|
||||
|
||||
if resp.get("b64_pdf"):
|
||||
with open(self.file_path, "wb") as f:
|
||||
f.write(base64.b64decode(resp["b64_pdf"]))
|
||||
|
||||
metadata["source"] = self.file_name
|
||||
doc = Document(page_content=content, metadata=metadata)
|
||||
return [doc]
|
||||
|
||||
|
||||
class ElemUnstructuredLoaderV0(BasePDFLoader):
|
||||
"""The appropriate parser is automatically selected based on the file format and OCR is supported"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
file_name: str,
|
||||
file_path: str,
|
||||
unstructured_api_key: str = None,
|
||||
unstructured_api_url: str = None,
|
||||
start: int = 0,
|
||||
n: int = None,
|
||||
verbose: bool = False,
|
||||
kwargs: dict = {},
|
||||
) -> None:
|
||||
"""Initialize with a file path."""
|
||||
self.unstructured_api_url = unstructured_api_url
|
||||
self.unstructured_api_key = unstructured_api_key
|
||||
self.start = start
|
||||
self.n = n
|
||||
self.headers = {"Content-Type": "application/json"}
|
||||
self.file_name = file_name
|
||||
self.extra_kwargs = kwargs
|
||||
super().__init__(file_path)
|
||||
|
||||
def load(self) -> List[Document]:
|
||||
page_content, metadata = self.get_text_metadata()
|
||||
doc = Document(page_content=page_content, metadata=metadata)
|
||||
return [doc]
|
||||
|
||||
def get_text_metadata(self):
|
||||
b64_data = base64.b64encode(open(self.file_path, "rb").read()).decode()
|
||||
payload = dict(
|
||||
filename=os.path.basename(self.file_name), b64_data=[b64_data], mode="text"
|
||||
)
|
||||
payload.update({"start": self.start, "n": self.n})
|
||||
payload.update(self.extra_kwargs)
|
||||
resp = requests.post(
|
||||
self.unstructured_api_url, headers=self.headers, json=payload
|
||||
)
|
||||
# 说明文件解析成功
|
||||
if resp.status_code == 200 and resp.json().get("status_code") == 200:
|
||||
res = resp.json()
|
||||
return res["text"], {"source": self.file_name}
|
||||
# 说明文件解析失败,pdf文件直接返回报错
|
||||
if self.file_name.endswith(".pdf"):
|
||||
raise Exception(
|
||||
f"file text {os.path.basename(self.file_name)} failed resp={resp.text}"
|
||||
)
|
||||
# 非pdf文件,先将文件转为pdf格式,让后再执行partition模式解析文档
|
||||
# 把文件转为pdf
|
||||
resp = requests.post(
|
||||
self.unstructured_api_url,
|
||||
headers=self.headers,
|
||||
json={
|
||||
"filename": os.path.basename(self.file_name),
|
||||
"b64_data": [b64_data],
|
||||
"mode": "topdf",
|
||||
},
|
||||
)
|
||||
if resp.status_code != 200 or resp.json().get("status_code") != 200:
|
||||
raise Exception(
|
||||
f"file topdf {os.path.basename(self.file_name)} failed resp={resp.text}"
|
||||
)
|
||||
# 解析pdf文件
|
||||
payload["mode"] = "partition"
|
||||
payload["b64_data"] = [resp.json()["b64_pdf"]]
|
||||
payload["filename"] = os.path.basename(self.file_name) + ".pdf"
|
||||
resp = requests.post(
|
||||
self.unstructured_api_url, headers=self.headers, json=payload
|
||||
)
|
||||
if resp.status_code != 200 or resp.json().get("status_code") != 200:
|
||||
raise Exception(
|
||||
f"file partition {os.path.basename(self.file_name)} failed resp={resp.text}"
|
||||
)
|
||||
res = resp.json()
|
||||
partitions = res["partitions"]
|
||||
if not partitions:
|
||||
raise Exception(
|
||||
f"file partition empty {os.path.basename(self.file_name)} resp={resp.text}"
|
||||
)
|
||||
# 拼接结果为文本
|
||||
content, _ = merge_partitions(self.file_path, partitions)
|
||||
return content, {"source": self.file_name}
|
||||
@@ -0,0 +1,350 @@
|
||||
import asyncio
|
||||
import os
|
||||
import io
|
||||
import json
|
||||
from typing import List
|
||||
|
||||
from bisheng.api.services.llm import LLMService
|
||||
from bisheng.utils import generate_uuid
|
||||
from fastapi import UploadFile, HTTPException
|
||||
import pandas as pd
|
||||
from collections import defaultdict
|
||||
from copy import deepcopy
|
||||
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.v1.schemas import (UnifiedResponseModel, resp_200, StreamData, BuildStatus)
|
||||
from bisheng.cache import InMemoryCache
|
||||
from bisheng.database.models.flow import FlowDao
|
||||
from bisheng.database.models.flow_version import FlowVersionDao
|
||||
from bisheng.database.models.assistant import AssistantDao
|
||||
from bisheng.api.services.flow import FlowService
|
||||
from bisheng.database.models.evaluation import (Evaluation, EvaluationDao, ExecType, EvaluationTaskStatus)
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.utils.minio_client import MinioClient
|
||||
from fastapi.encoders import jsonable_encoder
|
||||
from bisheng.utils.logger import logger
|
||||
from bisheng.api.services.assistant_agent import AssistantAgent
|
||||
from bisheng_ragas import evaluate
|
||||
from bisheng_ragas.llms.langchain import LangchainLLM
|
||||
from bisheng_ragas.metrics import AnswerCorrectnessBisheng
|
||||
from datasets import Dataset
|
||||
from bisheng_langchain.gpts.utils import import_by_type
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.api.utils import build_flow, build_input_keys_response
|
||||
from bisheng.graph.graph.base import Graph
|
||||
|
||||
flow_data_store = redis_client
|
||||
|
||||
expire = 600
|
||||
|
||||
|
||||
class EvaluationService:
|
||||
UserCache: InMemoryCache = InMemoryCache()
|
||||
|
||||
@classmethod
|
||||
def get_evaluation(cls,
|
||||
user: UserPayload,
|
||||
page: int = 1,
|
||||
limit: int = 20) -> UnifiedResponseModel[List[Evaluation]]:
|
||||
"""
|
||||
获取测评任务列表
|
||||
"""
|
||||
data = []
|
||||
res_evaluations, total = EvaluationDao.get_my_evaluations(user.user_id, page, limit)
|
||||
|
||||
# 技能ID列表
|
||||
flow_ids = []
|
||||
# 助手ID列表
|
||||
assistant_ids = []
|
||||
# 版本ID列表
|
||||
flow_version_ids = []
|
||||
|
||||
for one in res_evaluations:
|
||||
if one.exec_type == ExecType.FLOW.value:
|
||||
flow_ids.append(one.unique_id)
|
||||
if one.version:
|
||||
flow_version_ids.append(one.version)
|
||||
if one.exec_type == ExecType.ASSISTANT.value:
|
||||
assistant_ids.append(one.unique_id)
|
||||
|
||||
flow_names = {}
|
||||
flow_versions = {}
|
||||
assistant_names = {}
|
||||
|
||||
if flow_ids:
|
||||
flows = FlowDao.get_flow_by_ids(flow_ids=flow_ids)
|
||||
flow_names = {str(one.id): one.name for one in flows}
|
||||
|
||||
if flow_version_ids:
|
||||
versions = FlowVersionDao.get_list_by_ids(ids=flow_version_ids)
|
||||
flow_versions = {one.id: one.name for one in versions}
|
||||
|
||||
if assistant_ids:
|
||||
assistants = AssistantDao.get_assistants_by_ids(assistant_ids=assistant_ids)
|
||||
assistant_names = {str(one.id): one.name for one in assistants}
|
||||
|
||||
for one in res_evaluations:
|
||||
evaluation_item = jsonable_encoder(one)
|
||||
if one.exec_type == ExecType.FLOW.value:
|
||||
evaluation_item['unique_name'] = flow_names.get(one.unique_id)
|
||||
if one.exec_type == ExecType.ASSISTANT.value:
|
||||
evaluation_item['unique_name'] = assistant_names.get(one.unique_id)
|
||||
if one.version:
|
||||
evaluation_item['version_name'] = flow_versions.get(one.version)
|
||||
if one.result_score:
|
||||
evaluation_item['result_score'] = json.loads(one.result_score)
|
||||
|
||||
if one.status != EvaluationTaskStatus.running.value:
|
||||
evaluation_item['progress'] = f'100%'
|
||||
elif redis_client.exists(EvaluationService.get_redis_key(one.id)):
|
||||
evaluation_item['progress'] = f'{redis_client.get(EvaluationService.get_redis_key(one.id))}%'
|
||||
else:
|
||||
evaluation_item['progress'] = f'0%'
|
||||
|
||||
evaluation_item['user_name'] = cls.get_user_name(one.user_id)
|
||||
data.append(evaluation_item)
|
||||
|
||||
return resp_200(data={'data': data, 'total': total})
|
||||
|
||||
@classmethod
|
||||
def delete_evaluation(cls, evaluation_id: int, user_payload: UserPayload) -> UnifiedResponseModel:
|
||||
evaluation = EvaluationDao.get_user_one_evaluation(user_payload.user_id, evaluation_id)
|
||||
if not evaluation:
|
||||
raise HTTPException(status_code=404, detail='Evaluation not found')
|
||||
|
||||
EvaluationDao.delete_evaluation(evaluation)
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
def get_user_name(cls, user_id: int):
|
||||
if not user_id:
|
||||
return 'system'
|
||||
user = cls.UserCache.get(user_id)
|
||||
if user:
|
||||
return user.user_name
|
||||
user = UserDao.get_user(user_id)
|
||||
if not user:
|
||||
return f'{user_id}'
|
||||
cls.UserCache.set(user_id, user)
|
||||
return user.user_name
|
||||
|
||||
@classmethod
|
||||
def upload_file(cls, file: UploadFile):
|
||||
minio_client = MinioClient()
|
||||
file_id = generate_uuid()
|
||||
file_name = file.filename
|
||||
|
||||
file_ext = os.path.basename(file.filename).split('.')[-1]
|
||||
file_path = f'evaluation/dataset/{file_id}.{file_ext}'
|
||||
minio_client.upload_minio_file(file_path, file.file, content_type=file.content_type)
|
||||
return file_name, file_path
|
||||
|
||||
@classmethod
|
||||
def upload_result_file(cls, df: pd.DataFrame):
|
||||
minio_client = MinioClient()
|
||||
file_id = generate_uuid()
|
||||
|
||||
csv_buffer = io.BytesIO()
|
||||
df.to_csv(csv_buffer, index=False)
|
||||
csv_buffer.seek(0)
|
||||
|
||||
file_path = f'evaluation/result/{file_id}.csv'
|
||||
minio_client.upload_minio_data(object_name=file_path,
|
||||
data=csv_buffer.read(),
|
||||
length=csv_buffer.getbuffer().nbytes,
|
||||
content_type='application/csv')
|
||||
return file_path
|
||||
|
||||
@classmethod
|
||||
def read_csv_file(cls, file_path: str):
|
||||
minio_client = MinioClient()
|
||||
resp = minio_client.download_minio(file_path)
|
||||
if resp is None:
|
||||
return None
|
||||
new_data = io.BytesIO()
|
||||
for d in resp.stream(32 * 1024):
|
||||
new_data.write(d)
|
||||
resp.close()
|
||||
resp.release_conn()
|
||||
new_data.seek(0)
|
||||
return new_data
|
||||
|
||||
@classmethod
|
||||
def parse_csv(cls, file_data: io.BytesIO):
|
||||
df = pd.read_csv(file_data)
|
||||
df = df.dropna(axis=0, how='all').dropna(axis=1, how='all')
|
||||
if df.shape[1] < 2:
|
||||
raise ValueError("CSV file must have at least two columns")
|
||||
if df.columns[0] != 'question' or df.columns[1] != 'ground_truth':
|
||||
raise ValueError(
|
||||
"CSV file must have 'question' as the first column and 'ground_truth' as the second column")
|
||||
formatted_data = [{"question": row[0], "ground_truth": row[1]} for row in df.values]
|
||||
return formatted_data
|
||||
|
||||
@classmethod
|
||||
def get_redis_key(cls, evaluation_id: int):
|
||||
return f'evaluation_task_progress_{evaluation_id}'
|
||||
|
||||
@classmethod
|
||||
async def get_input_keys(cls, flow_id: int, version_id: int):
|
||||
artifacts = {}
|
||||
try:
|
||||
version_info = FlowVersionDao.get_version_by_id(version_id)
|
||||
if not version_info:
|
||||
return {"input": ""}
|
||||
|
||||
# L1 用户,采用build流程
|
||||
try:
|
||||
async for message in build_flow(graph_data=version_info.data,
|
||||
artifacts=artifacts,
|
||||
process_file=False,
|
||||
flow_id=flow_id,
|
||||
chat_id=None):
|
||||
if isinstance(message, Graph):
|
||||
graph = message
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'evaluation task get_input_keys {e}')
|
||||
return {"input": ""}
|
||||
|
||||
await graph.abuild()
|
||||
# Now we need to check the input_keys to send them to the client
|
||||
input_keys_response = {
|
||||
'input_keys': []
|
||||
}
|
||||
input_nodes = graph.get_input_nodes()
|
||||
for node in input_nodes:
|
||||
if hasattr(await node.get_result(), 'input_keys'):
|
||||
input_keys = build_input_keys_response(await node.get_result(), artifacts)
|
||||
input_keys['input_keys'].update({'id': node.id})
|
||||
input_keys_response['input_keys'].append(input_keys.get('input_keys'))
|
||||
elif 'fileNode' in node.output:
|
||||
input_keys_response['input_keys'].append({
|
||||
'file_path': '',
|
||||
'type': 'file',
|
||||
'id': node.id
|
||||
})
|
||||
if len(input_keys_response.get("input_keys")):
|
||||
input_item = input_keys_response.get("input_keys")[0]
|
||||
del input_item["id"]
|
||||
return input_item
|
||||
finally:
|
||||
pass
|
||||
return {"input": ""}
|
||||
|
||||
|
||||
def add_evaluation_task(evaluation_id: int):
|
||||
evaluation = EvaluationDao.get_one_evaluation(evaluation_id=evaluation_id)
|
||||
if not evaluation:
|
||||
return
|
||||
|
||||
redis_key = EvaluationService.get_redis_key(evaluation_id)
|
||||
|
||||
try:
|
||||
file_data = EvaluationService.read_csv_file(evaluation.file_path)
|
||||
csv_data = EvaluationService.parse_csv(file_data)
|
||||
progress_increment = 80 / len(csv_data)
|
||||
current_progress = 0
|
||||
|
||||
if evaluation.exec_type == ExecType.FLOW.value:
|
||||
flow_version = FlowVersionDao.get_version_by_id(version_id=evaluation.version)
|
||||
if not flow_version:
|
||||
raise Exception("Flow version not found")
|
||||
input_keys = asyncio.run(EvaluationService.get_input_keys(flow_id=evaluation.unique_id,
|
||||
version_id=evaluation.version))
|
||||
first_key = list(input_keys.keys())[0]
|
||||
|
||||
logger.info(f'evaluation task run flow input_keys: {input_keys} first_key: {first_key}')
|
||||
|
||||
for index, one in enumerate(csv_data):
|
||||
input_dict = deepcopy(input_keys)
|
||||
input_dict[first_key] = one.get('question')
|
||||
flow_index, flow_result = asyncio.run(FlowService.exec_flow_node(
|
||||
inputs=input_dict,
|
||||
tweaks={},
|
||||
index=0,
|
||||
versions=[flow_version]))
|
||||
one["answer"] = flow_result.get(flow_version.id)
|
||||
current_progress += progress_increment
|
||||
redis_client.set(redis_key, round(current_progress))
|
||||
|
||||
if evaluation.exec_type == ExecType.ASSISTANT.value:
|
||||
assistant = AssistantDao.get_one_assistant(evaluation.unique_id)
|
||||
if not assistant:
|
||||
raise Exception("Assistant not found")
|
||||
gpts_agent = AssistantAgent(assistant_info=assistant, chat_id="")
|
||||
asyncio.run(gpts_agent.init_assistant())
|
||||
for index, one in enumerate(csv_data):
|
||||
messages = asyncio.run(gpts_agent.run(one.get('question')))
|
||||
if len(messages):
|
||||
one["answer"] = messages[0].content
|
||||
current_progress += progress_increment
|
||||
redis_client.set(redis_key, round(current_progress))
|
||||
|
||||
_llm = LLMService.get_evaluation_llm_object()
|
||||
llm = LangchainLLM(_llm)
|
||||
data_samples = {
|
||||
"question": [one.get('question') for one in csv_data],
|
||||
"answer": [one.get('answer') for one in csv_data],
|
||||
"ground_truths": [[one.get('ground_truth')] for one in csv_data]
|
||||
}
|
||||
|
||||
dataset = Dataset.from_dict(data_samples)
|
||||
answer_correctness_bisheng = AnswerCorrectnessBisheng(llm=llm)
|
||||
score = evaluate(dataset, metrics=[answer_correctness_bisheng])
|
||||
df = score.to_pandas()
|
||||
result = df.to_dict(orient="list")
|
||||
logger.debug(f'evaluation id = {evaluation_id} result: {result}')
|
||||
|
||||
question = result.get('question', [])
|
||||
columns = [
|
||||
# 字段:标题:类型(1:文本 2:数字 3:百分比)
|
||||
("question", "question", 1),
|
||||
("ground_truths", "ground_truth", 1),
|
||||
("answer", "answer", 1),
|
||||
("statements_num_gt_only", "statements_num_gt_only", 2),
|
||||
("statements_num_answer_only", "statements_num_answer_only", 2),
|
||||
("statements_num_overlap", "statements_num_overlap", 2),
|
||||
("answer_recall", "recall", 3),
|
||||
("answer_precision", "precision", 3),
|
||||
("answer_f1", "F1", 3)
|
||||
]
|
||||
row_list = []
|
||||
tmp_dict = defaultdict(int)
|
||||
total_dict = {}
|
||||
|
||||
for index, one in enumerate(question):
|
||||
row_data = {}
|
||||
for field, title, unit_type in columns:
|
||||
value = result.get(field)[index]
|
||||
if unit_type != 1:
|
||||
tmp_dict[field] += value
|
||||
if unit_type == 3:
|
||||
value = f'{value * 100:.2f}%'
|
||||
row_data[title] = value
|
||||
row_list.append(row_data)
|
||||
|
||||
total_row_data = {}
|
||||
for field, title, unit_type in columns:
|
||||
value = tmp_dict.get(field)
|
||||
if unit_type == 3:
|
||||
value = f'{(value / len(row_list)) * 100:.2f}%'
|
||||
total_dict[field] = value
|
||||
total_row_data[title] = value
|
||||
row_list.append(total_row_data)
|
||||
|
||||
df = pd.DataFrame(data=row_list, columns=[one[1] for one in columns])
|
||||
result_file_path = EvaluationService.upload_result_file(df)
|
||||
|
||||
evaluation.result_score = json.dumps(total_dict)
|
||||
evaluation.status = EvaluationTaskStatus.success.value
|
||||
evaluation.result_file_path = result_file_path
|
||||
EvaluationDao.update_evaluation(evaluation=evaluation)
|
||||
redis_client.delete(redis_key)
|
||||
logger.info(f'evaluation task success id={evaluation_id}')
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f'evaluation task failed id={evaluation_id} {str(e)}')
|
||||
evaluation.status = EvaluationTaskStatus.failed.value
|
||||
EvaluationDao.update_evaluation(evaluation=evaluation)
|
||||
redis_client.delete(redis_key)
|
||||
@@ -3,28 +3,30 @@ import io
|
||||
import json
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import Any, Dict, List
|
||||
from uuid import UUID
|
||||
|
||||
from pydantic import ValidationError
|
||||
|
||||
from bisheng.api.errcode.finetune import (CancelJobError, ChangeModelNameError, CreateFinetuneError,
|
||||
DeleteJobError, ExportJobError, GetGPUInfoError,
|
||||
InvalidExtraParamsError, JobStatusError,
|
||||
ModelNameExistsError, NotFoundJobError,
|
||||
TrainDataNoneError, UnExportJobError)
|
||||
TrainDataNoneError, UnExportJobError, GetModelError)
|
||||
from bisheng.api.errcode.model_deploy import NotFoundModelError
|
||||
from bisheng.api.errcode.server import NoSftServerError
|
||||
from bisheng.api.services.rt_backend import RTBackend
|
||||
from bisheng.api.services.sft_backend import SFTBackend
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.utils import parse_gpus, parse_server_host
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel, resp_200
|
||||
from bisheng.cache import InMemoryCache
|
||||
from bisheng.database.models.finetune import (Finetune, FinetuneChangeModelName, FinetuneDao,
|
||||
FinetuneExtraParams, FinetuneList, FinetuneStatus)
|
||||
from bisheng.database.models.model_deploy import ModelDeploy, ModelDeployDao
|
||||
from bisheng.database.models.model_deploy import ModelDeploy, ModelDeployDao, ModelDeployInfo
|
||||
from bisheng.database.models.server import Server, ServerDao
|
||||
from bisheng.database.models.sft_model import SftModelDao
|
||||
from bisheng.utils.logger import logger
|
||||
from bisheng.utils.minio_client import MinioClient
|
||||
from pydantic import ValidationError
|
||||
|
||||
|
||||
sync_job_thread_pool = ThreadPoolExecutor(3)
|
||||
|
||||
@@ -158,13 +160,13 @@ class FinetuneService:
|
||||
finetune.root_model_name = root_model_name
|
||||
|
||||
# 调用SFT-backend的API新建任务
|
||||
logger.info(f'start create sft job: {finetune.id.hex}')
|
||||
logger.info(f'start create sft job: {finetune.id}')
|
||||
# 拼接指令所需的command参数
|
||||
command_params = cls.parse_command_params(finetune, base_model)
|
||||
sft_ret = SFTBackend.create_job(host=parse_server_host(finetune.sft_endpoint),
|
||||
job_id=finetune.id.hex, params=command_params)
|
||||
job_id=finetune.id, params=command_params)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'create sft job error: job_id: {finetune.id.hex}, err: {sft_ret[1]}')
|
||||
logger.error(f'create sft job error: job_id: {finetune.id}, err: {sft_ret[1]}')
|
||||
return CreateFinetuneError.return_resp()
|
||||
# 插入到数据库内
|
||||
FinetuneDao.insert_one(finetune)
|
||||
@@ -172,7 +174,7 @@ class FinetuneService:
|
||||
return resp_200(data=finetune)
|
||||
|
||||
@classmethod
|
||||
def cancel_job(cls, job_id: UUID, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
def cancel_job(cls, job_id: str, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
# 查看job任务信息
|
||||
finetune = FinetuneDao.find_job(job_id)
|
||||
if not finetune:
|
||||
@@ -186,7 +188,7 @@ class FinetuneService:
|
||||
|
||||
# 调用SFT-backend的API取消任务
|
||||
logger.info(f'start cancel job_id: {job_id}, user: {user.get("user_name")}')
|
||||
sft_ret = SFTBackend.cancel_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id.hex)
|
||||
sft_ret = SFTBackend.cancel_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'cancel sft job error: job_id: {job_id}, err: {sft_ret[1]}')
|
||||
return CancelJobError.return_resp()
|
||||
@@ -197,7 +199,7 @@ class FinetuneService:
|
||||
return resp_200(data=finetune)
|
||||
|
||||
@classmethod
|
||||
def delete_job(cls, job_id: UUID, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
def delete_job(cls, job_id: str, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
# 查看job任务信息
|
||||
finetune = FinetuneDao.find_job(job_id)
|
||||
if not finetune:
|
||||
@@ -207,7 +209,7 @@ class FinetuneService:
|
||||
|
||||
# 调用接口删除训练任务
|
||||
logger.info(f'start delete sft job: {job_id}, user: {user.get("user_name")}')
|
||||
sft_ret = SFTBackend.delete_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id.hex,
|
||||
sft_ret = SFTBackend.delete_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id,
|
||||
model_name=model_name)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'delete sft job error: job_id: {job_id}, err: {sft_ret[1]}')
|
||||
@@ -222,13 +224,13 @@ class FinetuneService:
|
||||
@classmethod
|
||||
def delete_job_log(cls, finetune: Finetune):
|
||||
minio_client = MinioClient()
|
||||
minio_client.delete_minio(f'/finetune/log/{finetune.id.hex}')
|
||||
minio_client.delete_minio(f'/finetune/log/{finetune.id}')
|
||||
|
||||
@classmethod
|
||||
def upload_job_log(cls, finetune: Finetune, log_data: io.BytesIO, length: int) -> str:
|
||||
minio_client = MinioClient()
|
||||
log_path = f'finetune/log/{finetune.id.hex}'
|
||||
minio_client.upload_minio_file(log_path, log_data, length)
|
||||
log_path = f'finetune/log/{finetune.id}'
|
||||
minio_client.upload_minio_file(log_path, log_data, length=length)
|
||||
return log_path
|
||||
|
||||
@classmethod
|
||||
@@ -272,7 +274,7 @@ class FinetuneService:
|
||||
return published_model.model
|
||||
|
||||
@classmethod
|
||||
def publish_job(cls, job_id: UUID, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
def publish_job(cls, job_id: str, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
# 查看job任务信息
|
||||
finetune = FinetuneDao.find_job(job_id)
|
||||
if not finetune:
|
||||
@@ -284,7 +286,7 @@ class FinetuneService:
|
||||
|
||||
# 调用SFT-backend的API接口
|
||||
logger.info(f'start export sft job: {job_id}, user: {user.get("user_name")}')
|
||||
sft_ret = SFTBackend.publish_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id.hex,
|
||||
sft_ret = SFTBackend.publish_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id,
|
||||
model_name=finetune.model_name)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'export sft job error: job_id: {job_id}, err: {sft_ret[1]}')
|
||||
@@ -313,7 +315,7 @@ class FinetuneService:
|
||||
return resp_200(data=finetune)
|
||||
|
||||
@classmethod
|
||||
def cancel_publish_job(cls, job_id: UUID, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
def cancel_publish_job(cls, job_id: str, user: Any) -> UnifiedResponseModel[Finetune]:
|
||||
# 查看job任务信息
|
||||
finetune = FinetuneDao.find_job(job_id)
|
||||
if not finetune:
|
||||
@@ -327,7 +329,7 @@ class FinetuneService:
|
||||
|
||||
# 调用SFT-backend的API接口
|
||||
logger.info(f'start cancel export sft job: {job_id}, user: {user.get("user_name")}')
|
||||
sft_ret = SFTBackend.cancel_publish_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id.hex,
|
||||
sft_ret = SFTBackend.cancel_publish_job(host=parse_server_host(finetune.sft_endpoint), job_id=job_id,
|
||||
model_name=finetune.model_name)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'cancel export sft job error: job_id: {job_id}, err: {sft_ret[1]}')
|
||||
@@ -369,7 +371,7 @@ class FinetuneService:
|
||||
cls.sync_job_status(finetune, finetune.sft_endpoint)
|
||||
|
||||
@classmethod
|
||||
def get_job_info(cls, job_id: UUID) -> UnifiedResponseModel:
|
||||
def get_job_info(cls, job_id: str) -> UnifiedResponseModel:
|
||||
""" 获取训练中任务的实时信息 """
|
||||
# 查看job任务信息
|
||||
finetune = FinetuneDao.find_job(job_id)
|
||||
@@ -405,7 +407,7 @@ class FinetuneService:
|
||||
sub_data = {'step': None, 'loss': None}
|
||||
elem = elem.strip()
|
||||
elem_data = json.loads(elem)
|
||||
if elem_data['loss'] is None:
|
||||
if elem_data.get('loss', None) is None:
|
||||
continue
|
||||
sub_data['step'] = elem_data['current_steps']
|
||||
sub_data['loss'] = elem_data['loss']
|
||||
@@ -417,11 +419,11 @@ class FinetuneService:
|
||||
""" 从SFT-backend服务同步任务状态 """
|
||||
if finetune.status != FinetuneStatus.TRAINING.value:
|
||||
return True
|
||||
logger.info(f'start sync job status: {finetune.id.hex}')
|
||||
logger.info(f'start sync job status: {finetune.id}')
|
||||
|
||||
sft_ret = SFTBackend.get_job_status(host=parse_server_host(sft_endpoint), job_id=finetune.id.hex)
|
||||
sft_ret = SFTBackend.get_job_status(host=parse_server_host(sft_endpoint), job_id=finetune.id)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'get sft job status error: job_id: {finetune.id.hex}, err: {sft_ret[1]}')
|
||||
logger.error(f'get sft job status error: job_id: {finetune.id}, err: {sft_ret[1]}')
|
||||
return False
|
||||
if sft_ret[1]['status'] == SFTBackend.JOB_FINISHED:
|
||||
finetune.status = FinetuneStatus.SUCCESS.value
|
||||
@@ -438,9 +440,9 @@ class FinetuneService:
|
||||
|
||||
# 查询任务执行日志和报告
|
||||
logger.info('start query sft job log and report')
|
||||
sft_ret = SFTBackend.get_job_log(host=parse_server_host(sft_endpoint), job_id=finetune.id.hex)
|
||||
sft_ret = SFTBackend.get_job_log(host=parse_server_host(sft_endpoint), job_id=finetune.id)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'get sft job log error: job_id: {finetune.id.hex}, err: {sft_ret[1]}')
|
||||
logger.error(f'get sft job log error: job_id: {finetune.id}, err: {sft_ret[1]}')
|
||||
log_data = sft_ret[1]['log_data'].encode('utf-8')
|
||||
# 上传日志文件到minio上
|
||||
log_path = cls.upload_job_log(finetune, io.BytesIO(log_data), len(log_data))
|
||||
@@ -448,9 +450,9 @@ class FinetuneService:
|
||||
|
||||
# 查询任务评估报告
|
||||
logger.info('start query sft job report')
|
||||
sft_ret = SFTBackend.get_job_metrics(host=parse_server_host(sft_endpoint), job_id=finetune.id.hex)
|
||||
sft_ret = SFTBackend.get_job_metrics(host=parse_server_host(sft_endpoint), job_id=finetune.id)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'get sft job report error: job_id: {finetune.id.hex}, err: {sft_ret[1]}')
|
||||
logger.error(f'get sft job report error: job_id: {finetune.id}, err: {sft_ret[1]}')
|
||||
else:
|
||||
finetune.report = sft_ret[1]['report']
|
||||
|
||||
@@ -487,14 +489,14 @@ class FinetuneService:
|
||||
return True
|
||||
published_model = ModelDeployDao.find_model(finetune.model_id)
|
||||
if not published_model:
|
||||
logger.error(f'published model not found, job_id: {finetune.id.hex}, model_id: {finetune.model_id}')
|
||||
logger.error(f'published model not found, job_id: {finetune.id}, model_id: {finetune.model_id}')
|
||||
return False
|
||||
|
||||
# 调用接口修改已发布模型的名称
|
||||
sft_ret = SFTBackend.change_model_name(parse_server_host(finetune.sft_endpoint), finetune.id.hex,
|
||||
sft_ret = SFTBackend.change_model_name(parse_server_host(finetune.sft_endpoint), finetune.id,
|
||||
published_model.model, model_name)
|
||||
if not sft_ret[0]:
|
||||
logger.error(f'change model name error: job_id: {finetune.id.hex}, err: {sft_ret[1]}')
|
||||
logger.error(f'change model name error: job_id: {finetune.id}, err: {sft_ret[1]}')
|
||||
return False
|
||||
|
||||
# 修改可预训练的模型名称
|
||||
@@ -507,7 +509,7 @@ class FinetuneService:
|
||||
|
||||
@classmethod
|
||||
def get_server_filters(cls) -> UnifiedResponseModel:
|
||||
""" 获取服务器过滤条件 """
|
||||
""" 获取ft服务器过滤条件 """
|
||||
server_filters = FinetuneDao.get_server_filters()
|
||||
res = []
|
||||
for one in server_filters:
|
||||
@@ -517,6 +519,32 @@ class FinetuneService:
|
||||
})
|
||||
return resp_200(data=res)
|
||||
|
||||
@classmethod
|
||||
def get_model_list(cls, login_user: UserPayload, server_id: int) -> List[ModelDeploy]:
|
||||
""" 获取ft服务下的所有模型列表 """
|
||||
server_info = ServerDao.find_server(server_id)
|
||||
if not server_info:
|
||||
raise NoSftServerError.http_exception()
|
||||
flag, model_name_list = SFTBackend.get_all_model(parse_server_host(server_info.sft_endpoint))
|
||||
if not flag:
|
||||
logger.error(f'get model list error: server_id: {server_id}, err: {model_name_list}')
|
||||
raise GetModelError.http_exception()
|
||||
ret = []
|
||||
db_model = ModelDeployDao.find_model_by_server(str(server_id))
|
||||
for one in db_model:
|
||||
if one.model in model_name_list:
|
||||
ret.append(one)
|
||||
model_name_list.remove(one.model)
|
||||
for one in model_name_list:
|
||||
ret.append(ModelDeployDao.insert_one(ModelDeploy(server=str(server_id),
|
||||
model=one,
|
||||
endpoint=f'http://{server_info.endpoint}/v2.1/models')))
|
||||
|
||||
res = []
|
||||
for one in ret:
|
||||
res.append(ModelDeployInfo(**one.dict(), sft_support=True))
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
def get_gpu_info(cls) -> UnifiedResponseModel:
|
||||
""" 获取GPU信息 """
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
import os.path
|
||||
import uuid
|
||||
from typing import Any, List
|
||||
|
||||
from bisheng.api.errcode.finetune import TrainFileNotExistError
|
||||
from bisheng.api.v1.schema.base_schema import PageList
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel, resp_200
|
||||
from bisheng.database.models.preset_train import PresetTrain, PresetTrainDao
|
||||
from bisheng.utils import generate_uuid
|
||||
from bisheng.utils.logger import logger
|
||||
from bisheng.utils.minio_client import MinioClient
|
||||
from fastapi import UploadFile
|
||||
@@ -15,7 +16,8 @@ class FinetuneFileService(BaseModel):
|
||||
""" 训练任务 文件管理 """
|
||||
|
||||
@classmethod
|
||||
def upload_file(cls, files: List[UploadFile], is_preset: bool, user: Any) -> UnifiedResponseModel:
|
||||
def upload_file(cls, files: List[UploadFile], is_preset: bool,
|
||||
user: Any) -> UnifiedResponseModel:
|
||||
if len(files) == 0:
|
||||
return TrainFileNotExistError.return_resp()
|
||||
|
||||
@@ -28,6 +30,25 @@ class FinetuneFileService(BaseModel):
|
||||
PresetTrainDao.insert_batch(file_list)
|
||||
return resp_200(data=file_list)
|
||||
|
||||
@classmethod
|
||||
def upload_preset_file(cls, name: str, type: int, file_path: str,
|
||||
user: Any) -> UnifiedResponseModel:
|
||||
# 将训练文件上传到minio
|
||||
file_root = cls.get_upload_file_root(False)
|
||||
file_id = generate_uuid()
|
||||
file_ext = os.path.basename(file_path).split('.')[-1]
|
||||
object_name = f'{file_root}/{file_id}.{file_ext}'
|
||||
MinioClient().upload_minio(object_name, file_path)
|
||||
# 将预置数据存入数据库
|
||||
file_info = PresetTrain(id=file_id,
|
||||
name=name,
|
||||
url=object_name,
|
||||
type=type,
|
||||
user_id=user.get('user_id'),
|
||||
user_name=user.get('user_name'))
|
||||
PresetTrainDao.insert_batch([file_info])
|
||||
return resp_200(data=file_info)
|
||||
|
||||
@classmethod
|
||||
def get_upload_file_root(cls, is_preset: bool) -> str:
|
||||
if is_preset:
|
||||
@@ -36,25 +57,35 @@ class FinetuneFileService(BaseModel):
|
||||
return 'finetune/train_file/personal'
|
||||
|
||||
@classmethod
|
||||
def upload_file_to_minio(cls, files: List[UploadFile], file_root: str, user: Any) -> List[PresetTrain]:
|
||||
def upload_file_to_minio(cls, files: List[UploadFile], file_root: str,
|
||||
user: Any) -> List[PresetTrain]:
|
||||
minio_client = MinioClient()
|
||||
ret = []
|
||||
for file in files:
|
||||
file_id = uuid.uuid4().hex
|
||||
file_id = generate_uuid()
|
||||
file_ext = os.path.basename(file.filename).split('.')[-1]
|
||||
file_info = PresetTrain(id=file_id, name=file.filename,
|
||||
file_info = PresetTrain(id=file_id,
|
||||
name=file.filename,
|
||||
url=f'{file_root}/{file_id}.{file_ext}',
|
||||
user_id=user.get('user_id'), user_name=user.get('user_name'))
|
||||
minio_client.upload_minio_file(file_info.url, file.file, file.size, content_type=file.content_type)
|
||||
user_id=user.get('user_id'),
|
||||
user_name=user.get('user_name'))
|
||||
minio_client.upload_minio_file(file_info.url,
|
||||
file.file,
|
||||
length=file.size,
|
||||
content_type=file.content_type)
|
||||
ret.append(file_info)
|
||||
return ret
|
||||
|
||||
@classmethod
|
||||
def get_preset_file(cls) -> List[PresetTrain]:
|
||||
return PresetTrainDao.find_all()
|
||||
def get_preset_file(cls,
|
||||
keyword: str = None,
|
||||
page_size: int = None,
|
||||
page_num: int = None) -> List[PresetTrain]:
|
||||
list_res, total_count = PresetTrainDao.search_name(keyword, page_size, page_num)
|
||||
return PageList(list=list_res, total=total_count)
|
||||
|
||||
@classmethod
|
||||
def delete_preset_file(cls, file_id: uuid.UUID, user: Any) -> UnifiedResponseModel:
|
||||
def delete_preset_file(cls, file_id: str, user: Any) -> UnifiedResponseModel:
|
||||
file_data = PresetTrainDao.find_one(file_id)
|
||||
if not file_data:
|
||||
return TrainFileNotExistError.return_resp()
|
||||
|
||||
@@ -1,29 +1,35 @@
|
||||
import asyncio
|
||||
import copy
|
||||
from typing import List, Dict, AsyncGenerator
|
||||
from typing import List, Dict, AsyncGenerator, Optional
|
||||
|
||||
from fastapi.encoders import jsonable_encoder
|
||||
from fastapi import Request
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.errcode.base import UnAuthorizedError
|
||||
from bisheng.api.errcode.base import UnAuthorizedError, NotFoundError
|
||||
from bisheng.api.errcode.flow import NotFoundVersionError, CurVersionDelError, VersionNameExistsError, \
|
||||
NotFoundFlowError, \
|
||||
FlowOnlineEditError
|
||||
FlowOnlineEditError, WorkFlowOnlineEditError
|
||||
from bisheng.api.services.audit_log import AuditLogService
|
||||
from bisheng.api.services.base import BaseService
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.utils import get_L2_param_from_flow
|
||||
from bisheng.api.utils import get_L2_param_from_flow, get_request_ip
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel, resp_200, FlowVersionCreate, FlowCompareReq, resp_500, \
|
||||
StreamData
|
||||
from bisheng.chat.utils import process_node_data
|
||||
from bisheng.database.models.flow import FlowDao, FlowStatus
|
||||
from bisheng.database.models.flow import FlowDao, FlowStatus, Flow, FlowType
|
||||
from bisheng.database.models.flow_version import FlowVersionDao, FlowVersionRead, FlowVersion
|
||||
from bisheng.database.models.group_resource import GroupResourceDao, ResourceTypeEnum, GroupResource
|
||||
from bisheng.database.models.role_access import RoleAccessDao, AccessType
|
||||
from bisheng.database.models.tag import TagDao
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.database.models.user_group import UserGroupDao
|
||||
from bisheng.database.models.user_role import UserRoleDao
|
||||
from bisheng.database.models.variable_value import VariableDao
|
||||
from bisheng.processing.process import process_graph_cached, process_tweaks
|
||||
|
||||
|
||||
class FlowService:
|
||||
class FlowService(BaseService):
|
||||
|
||||
@classmethod
|
||||
def get_version_list_by_flow(cls, user: UserPayload, flow_id: str) -> UnifiedResponseModel[List[FlowVersionRead]]:
|
||||
@@ -59,8 +65,12 @@ class FlowService:
|
||||
if not flow_info:
|
||||
return NotFoundFlowError.return_resp()
|
||||
|
||||
atype = AccessType.FLOW_WRITE
|
||||
if flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
atype = AccessType.WORK_FLOW_WRITE
|
||||
|
||||
# 判断权限
|
||||
if not user.access_check(flow_info.user_id, flow_info.id.hex, AccessType.FLOW_WRITE):
|
||||
if not user.access_check(flow_info.user_id, flow_info.id, atype):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
if version_info.is_current == 1:
|
||||
@@ -70,7 +80,8 @@ class FlowService:
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
def change_current_version(cls, user: UserPayload, flow_id: str, version_id: int) -> UnifiedResponseModel[None]:
|
||||
def change_current_version(cls, request: Request, login_user: UserPayload, flow_id: str, version_id: int) \
|
||||
-> UnifiedResponseModel[None]:
|
||||
"""
|
||||
修改当前版本
|
||||
"""
|
||||
@@ -78,8 +89,12 @@ class FlowService:
|
||||
if not flow_info:
|
||||
return NotFoundFlowError.return_resp()
|
||||
|
||||
atype = AccessType.FLOW_WRITE
|
||||
if flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
atype = AccessType.WORK_FLOW_WRITE
|
||||
|
||||
# 判断权限
|
||||
if not user.access_check(flow_info.user_id, flow_info.id.hex, AccessType.FLOW_WRITE):
|
||||
if not login_user.access_check(flow_info.user_id, flow_info.id, atype):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
# 技能上线状态不允许 切换版本
|
||||
@@ -95,6 +110,8 @@ class FlowService:
|
||||
|
||||
# 修改当前版本为用户选择的版本
|
||||
FlowVersionDao.change_current_version(flow_id, version_info)
|
||||
|
||||
cls.update_flow_hook(request, login_user, flow_info)
|
||||
return resp_200()
|
||||
|
||||
@classmethod
|
||||
@@ -108,7 +125,7 @@ class FlowService:
|
||||
return NotFoundFlowError.return_resp()
|
||||
|
||||
# 判断权限
|
||||
if not user.access_check(flow_info.user_id, flow_info.id.hex, AccessType.FLOW_WRITE):
|
||||
if not user.access_check(flow_info.user_id, flow_info.id, AccessType.FLOW_WRITE):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
exist_version = FlowVersionDao.get_version_by_name(flow_id, flow_version.name)
|
||||
@@ -117,7 +134,8 @@ class FlowService:
|
||||
|
||||
flow_version = FlowVersion(flow_id=flow_id, name=flow_version.name, description=flow_version.description,
|
||||
user_id=user.user_id, data=flow_version.data,
|
||||
original_version_id=flow_version.original_version_id)
|
||||
original_version_id=flow_version.original_version_id,
|
||||
flow_type=flow_version.flow_type)
|
||||
|
||||
# 创建新版本
|
||||
flow_version = FlowVersionDao.create_version(flow_version)
|
||||
@@ -135,7 +153,7 @@ class FlowService:
|
||||
return resp_200(data=flow_version)
|
||||
|
||||
@classmethod
|
||||
def update_version_info(cls, user: UserPayload, version_id: int, flow_version: FlowVersionCreate) \
|
||||
def update_version_info(cls, request: Request, user: UserPayload, version_id: int, flow_version: FlowVersionCreate) \
|
||||
-> UnifiedResponseModel[FlowVersion]:
|
||||
"""
|
||||
更新版本信息
|
||||
@@ -148,13 +166,19 @@ class FlowService:
|
||||
if not flow_info:
|
||||
return NotFoundFlowError.return_resp()
|
||||
|
||||
atype = AccessType.FLOW_WRITE
|
||||
if flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
atype = AccessType.WORK_FLOW_WRITE
|
||||
# 判断权限
|
||||
if not user.access_check(flow_info.user_id, flow_info.id.hex, AccessType.FLOW_WRITE):
|
||||
if not user.access_check(flow_info.user_id, flow_info.id, atype):
|
||||
return UnAuthorizedError.return_resp()
|
||||
|
||||
# 版本是当前版本, 且技能处于上线状态则不可编辑
|
||||
if version_info.is_current == 1 and flow_info.status == FlowStatus.ONLINE.value:
|
||||
return FlowOnlineEditError.return_resp()
|
||||
# 版本是当前版本, 且技能处于上线状态则不可编辑data数据,名称和描述可以编辑
|
||||
if version_info.is_current == 1 and flow_info.status == FlowStatus.ONLINE.value and flow_version.data:
|
||||
if flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
return WorkFlowOnlineEditError.return_resp()
|
||||
else:
|
||||
return FlowOnlineEditError.return_resp()
|
||||
|
||||
version_info.name = flow_version.name if flow_version.name else version_info.name
|
||||
version_info.description = flow_version.description if flow_version.description else version_info.description
|
||||
@@ -164,33 +188,63 @@ class FlowService:
|
||||
|
||||
flow_version = FlowVersionDao.update_version(version_info)
|
||||
|
||||
try:
|
||||
# 重新整理此版本的表单数据
|
||||
if not get_L2_param_from_flow(flow_version.data, flow_version.flow_id, flow_version.id):
|
||||
logger.error(f'flow_id={flow_version.id} version_id={flow_version.id} extract file_node fail')
|
||||
except:
|
||||
pass
|
||||
if flow_version.flow_type == FlowType.FLOW.value:
|
||||
try:
|
||||
# 重新整理此版本的表单数据
|
||||
if not get_L2_param_from_flow(flow_version.data, flow_version.flow_id, flow_version.id):
|
||||
logger.error(f'flow_id={flow_version.id} version_id={flow_version.id} extract file_node fail')
|
||||
except:
|
||||
pass
|
||||
cls.update_flow_hook(request, user, flow_info)
|
||||
return resp_200(data=flow_version)
|
||||
|
||||
@classmethod
|
||||
def get_all_flows(cls, user: UserPayload, name: str, status: int, page: int = 1, page_size: int = 10) -> \
|
||||
UnifiedResponseModel[List[Dict]]:
|
||||
def get_one_flow(cls, login_user: UserPayload, flow_id: str) -> UnifiedResponseModel[Flow]:
|
||||
"""
|
||||
获取单个技能的详情
|
||||
"""
|
||||
flow_info = FlowDao.get_flow_by_id(flow_id)
|
||||
if not flow_info:
|
||||
raise NotFoundFlowError.http_exception()
|
||||
atype = AccessType.FLOW
|
||||
if flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
atype = AccessType.WORK_FLOW
|
||||
if not login_user.access_check(flow_info.user_id, flow_info.id, atype):
|
||||
raise UnAuthorizedError.http_exception()
|
||||
flow_info.logo = cls.get_logo_share_link(flow_info.logo)
|
||||
|
||||
return resp_200(data=flow_info)
|
||||
|
||||
@classmethod
|
||||
def get_all_flows(cls, user: UserPayload, name: str, status: int, tag_id: int = 0, page: int = 1,
|
||||
page_size: int = 10, flow_type: Optional[int] = FlowType.FLOW.value) -> UnifiedResponseModel[
|
||||
List[Dict]]:
|
||||
"""
|
||||
获取所有技能
|
||||
"""
|
||||
flow_ids = []
|
||||
if tag_id:
|
||||
ret = TagDao.get_resources_by_tags_batch([tag_id], [ResourceTypeEnum.FLOW,ResourceTypeEnum.WORK_FLOW])
|
||||
flow_ids = [one.resource_id for one in ret]
|
||||
assistant_ids = [one.resource_id for one in ret]
|
||||
if not assistant_ids:
|
||||
return resp_200(data={
|
||||
'data': [],
|
||||
'total': 0
|
||||
})
|
||||
# 获取用户可见的技能列表
|
||||
if user.is_admin():
|
||||
data = FlowDao.get_flows(user.user_id, "admin", name, status, page, page_size)
|
||||
total = FlowDao.count_flows(user.user_id, "admin", name, status)
|
||||
data = FlowDao.get_flows(user.user_id, "admin", name, status, flow_ids, page, page_size, flow_type)
|
||||
total = FlowDao.count_flows(user.user_id, "admin", name, status, flow_ids, flow_type)
|
||||
else:
|
||||
user_role = UserRoleDao.get_user_roles(user.user_id)
|
||||
role_ids = [role.role_id for role in user_role]
|
||||
role_access = RoleAccessDao.get_role_access(role_ids, AccessType.FLOW)
|
||||
role_access = RoleAccessDao.get_role_access_batch(role_ids, [AccessType.FLOW,AccessType.WORK_FLOW])
|
||||
flow_id_extra = []
|
||||
if role_access:
|
||||
flow_id_extra = [access.third_id for access in role_access]
|
||||
data = FlowDao.get_flows(user.user_id, flow_id_extra, name, status, page, page_size)
|
||||
total = FlowDao.count_flows(user.user_id, flow_id_extra, name, status)
|
||||
data = FlowDao.get_flows(user.user_id, flow_id_extra, name, status, flow_ids, page, page_size, flow_type)
|
||||
total = FlowDao.count_flows(user.user_id, flow_id_extra, name, status, flow_ids, flow_type)
|
||||
|
||||
# 获取技能列表对应的用户信息和版本信息
|
||||
# 技能ID列表
|
||||
@@ -198,7 +252,7 @@ class FlowService:
|
||||
# 技能创建用户的ID列表
|
||||
user_ids = []
|
||||
for one in data:
|
||||
flow_ids.append(one.id.hex)
|
||||
flow_ids.append(one.id)
|
||||
user_ids.append(one.user_id)
|
||||
# 获取列表内的用户信息
|
||||
user_infos = UserDao.get_user_by_ids(user_ids)
|
||||
@@ -212,13 +266,28 @@ class FlowService:
|
||||
flow_versions[one.flow_id] = []
|
||||
flow_versions[one.flow_id].append(jsonable_encoder(one))
|
||||
|
||||
# 获取技能所属的分组
|
||||
flow_groups = GroupResourceDao.get_resources_group(ResourceTypeEnum.FLOW, flow_ids)
|
||||
flow_group_dict = {}
|
||||
for one in flow_groups:
|
||||
if one.third_id not in flow_group_dict:
|
||||
flow_group_dict[one.third_id] = []
|
||||
flow_group_dict[one.third_id].append(one.group_id)
|
||||
|
||||
# 获取技能关联的tag
|
||||
flow_tags = TagDao.get_tags_by_resource(ResourceTypeEnum.FLOW, flow_ids)
|
||||
|
||||
# 重新拼接技能列表list信息
|
||||
res = []
|
||||
for one in data:
|
||||
one.logo = cls.get_logo_share_link(one.logo)
|
||||
flow_info = jsonable_encoder(one)
|
||||
flow_info['user_name'] = user_dict.get(one.user_id, one.user_id)
|
||||
flow_info['write'] = True if user.is_admin() or user.user_id == one.user_id else False
|
||||
flow_info['version_list'] = flow_versions.get(one.id.hex, [])
|
||||
flow_info['version_list'] = flow_versions.get(one.id, [])
|
||||
flow_info['group_ids'] = flow_group_dict.get(one.id, [])
|
||||
flow_info['tags'] = flow_tags.get(one.id, [])
|
||||
|
||||
res.append(flow_info)
|
||||
|
||||
return resp_200(data={
|
||||
@@ -342,3 +411,55 @@ class FlowService:
|
||||
answer_result[one.id] = list(task_result.values())[0]
|
||||
|
||||
return index, answer_result
|
||||
|
||||
@classmethod
|
||||
def create_flow_hook(cls, request: Request, login_user: UserPayload, flow_info: Flow, version_id,
|
||||
flow_type: Optional[int] = None) -> bool:
|
||||
logger.info(f'create_flow_hook flow: {flow_info.id}, user_payload: {login_user.user_id}')
|
||||
# 将技能所需的表单写到数据库内
|
||||
try:
|
||||
if flow_info.data and not get_L2_param_from_flow(flow_info.data, flow_info.id, version_id):
|
||||
logger.error(f'flow_id={flow_info.id} extract file_node fail')
|
||||
except Exception:
|
||||
pass
|
||||
# 将技能关联到对应的用户组下
|
||||
user_group = UserGroupDao.get_user_group(login_user.user_id)
|
||||
if user_group:
|
||||
batch_resource = []
|
||||
resource_type = ResourceTypeEnum.FLOW.value
|
||||
if flow_type and flow_type == FlowType.WORKFLOW.value:
|
||||
resource_type = ResourceTypeEnum.WORK_FLOW.value
|
||||
|
||||
for one in user_group:
|
||||
batch_resource.append(
|
||||
GroupResource(group_id=one.group_id,
|
||||
third_id=flow_info.id,
|
||||
type=resource_type))
|
||||
GroupResourceDao.insert_group_batch(batch_resource)
|
||||
# 写入审计日志
|
||||
AuditLogService.create_build_flow(login_user, get_request_ip(request), flow_info.id, flow_type)
|
||||
|
||||
# 写入logo缓存
|
||||
cls.get_logo_share_link(flow_info.logo)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def update_flow_hook(cls, request: Request, login_user: UserPayload, flow_info: Flow) -> bool:
|
||||
# 写入审计日志
|
||||
AuditLogService.update_build_flow(login_user, get_request_ip(request), flow_info.id,
|
||||
flow_type=flow_info.flow_type)
|
||||
|
||||
# 写入logo缓存
|
||||
cls.get_logo_share_link(flow_info.logo)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def delete_flow_hook(cls, request: Request, login_user: UserPayload, flow_info: Flow) -> bool:
|
||||
logger.info(f'delete_flow_hook flow: {flow_info.id}, user_payload: {login_user.user_id}')
|
||||
|
||||
# 写入审计日志
|
||||
AuditLogService.delete_build_flow(login_user, get_request_ip(request), flow_info, flow_type=flow_info.flow_type)
|
||||
|
||||
# 将用户组下关联的技能删除
|
||||
GroupResourceDao.delete_group_resource_by_third_id(flow_info.id, ResourceTypeEnum.FLOW)
|
||||
return True
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
"""
|
||||
@project: qabot
|
||||
@Author:虎
|
||||
@file: __init__.py.py
|
||||
@date:2023/9/6 10:09
|
||||
@desc:
|
||||
"""
|
||||
@@ -0,0 +1,23 @@
|
||||
"""
|
||||
@project: maxkb
|
||||
@Author:虎
|
||||
@file: base_parse_qa_handle.py
|
||||
@date:2024/5/21 14:56
|
||||
@desc:
|
||||
"""
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
|
||||
class BaseParseTableHandle(ABC):
|
||||
|
||||
@abstractmethod
|
||||
def support(self, file, get_buffer):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def handle(self, file, get_buffer, save_image):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_content(self, file, save_image):
|
||||
pass
|
||||
@@ -0,0 +1,23 @@
|
||||
"""
|
||||
@project: maxkb
|
||||
@Author:虎
|
||||
@file: base_split_handle.py
|
||||
@date:2024/3/27 18:13
|
||||
@desc:
|
||||
"""
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import List
|
||||
|
||||
|
||||
class BaseSplitHandle(ABC):
|
||||
@abstractmethod
|
||||
def support(self, file, get_buffer):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def handle(self, file, pattern_list: List, with_filter: bool, limit: int, get_buffer, save_image):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_content(self, file, save_image):
|
||||
pass
|
||||
@@ -0,0 +1,45 @@
|
||||
# import logging
|
||||
from bisheng.api.services.handler.base_parse_table_handle import BaseParseTableHandle
|
||||
from charset_normalizer import detect
|
||||
from loguru import logger as max_kb
|
||||
|
||||
# from common.handle.base_parse_table_handle import BaseParseTableHandle
|
||||
|
||||
# max_kb = logging.getLogger("max_kb")
|
||||
|
||||
|
||||
class CsvSplitHandle(BaseParseTableHandle):
|
||||
|
||||
def support(self, file, get_buffer):
|
||||
file_name: str = file.name.lower()
|
||||
if file_name.endswith('.csv'):
|
||||
return True
|
||||
return False
|
||||
|
||||
def handle(self, file, get_buffer, save_image):
|
||||
buffer = get_buffer(file)
|
||||
try:
|
||||
content = buffer.decode(detect(buffer)['encoding'])
|
||||
except BaseException as e:
|
||||
max_kb.error(f'csv split handle error: {e}')
|
||||
return [{'name': file.name, 'paragraphs': []}]
|
||||
|
||||
csv_model = content.split('\n')
|
||||
paragraphs = []
|
||||
# 第一行为标题
|
||||
title = csv_model[0].split(',')
|
||||
for row in csv_model[1:]:
|
||||
if not row:
|
||||
continue
|
||||
line = '; '.join([f'{key}:{value}' for key, value in zip(title, row.split(','))])
|
||||
paragraphs.append({'title': '', 'content': line})
|
||||
|
||||
return [{'name': file.name, 'paragraphs': paragraphs}]
|
||||
|
||||
def get_content(self, file, save_image):
|
||||
buffer = file.read()
|
||||
try:
|
||||
return buffer.decode(detect(buffer)['encoding'])
|
||||
except BaseException as e:
|
||||
max_kb.error(f'csv split handle error: {e}')
|
||||
return f'error: {e}'
|
||||
@@ -0,0 +1,94 @@
|
||||
# import logging
|
||||
|
||||
import xlrd
|
||||
from loguru import logger as max_kb
|
||||
|
||||
from bisheng.api.services.handler.base_parse_table_handle import BaseParseTableHandle
|
||||
|
||||
|
||||
# from common.handle.base_parse_table_handle import BaseParseTableHandle
|
||||
|
||||
# max_kb = logging.getLogger("max_kb")
|
||||
|
||||
|
||||
class XlsSplitHandle(BaseParseTableHandle):
|
||||
|
||||
def support(self, file, get_buffer):
|
||||
file_name: str = file.name.lower()
|
||||
buffer = get_buffer(file)
|
||||
if file_name.endswith('.xls') and xlrd.inspect_format(content=buffer):
|
||||
return True
|
||||
return False
|
||||
|
||||
def handle(self, file, get_buffer, save_image):
|
||||
buffer = get_buffer(file)
|
||||
try:
|
||||
wb = xlrd.open_workbook(file_contents=buffer, formatting_info=True)
|
||||
result = []
|
||||
sheets = wb.sheets()
|
||||
for sheet in sheets:
|
||||
# 获取合并单元格的范围信息
|
||||
merged_cells = sheet.merged_cells
|
||||
data = []
|
||||
paragraphs = []
|
||||
# 获取第一行作为标题行
|
||||
headers = [sheet.cell_value(0, col_idx) for col_idx in range(sheet.ncols)]
|
||||
# 从第二行开始遍历每一行(跳过标题行)
|
||||
for row_idx in range(1, sheet.nrows):
|
||||
row_data = {}
|
||||
for col_idx in range(sheet.ncols):
|
||||
cell_value = sheet.cell_value(row_idx, col_idx)
|
||||
|
||||
# 检查是否为空单元格,如果为空检查是否在合并区域中
|
||||
if cell_value == '':
|
||||
# 检查当前单元格是否在合并区域
|
||||
for (rlo, rhi, clo, chi) in merged_cells:
|
||||
if rlo <= row_idx < rhi and clo <= col_idx < chi:
|
||||
# 使用合并区域的左上角单元格的值
|
||||
cell_value = sheet.cell_value(rlo, clo)
|
||||
break
|
||||
|
||||
# 将标题作为键,单元格的值作为值存入字典
|
||||
row_data[headers[col_idx]] = cell_value
|
||||
data.append(row_data)
|
||||
|
||||
for row in data:
|
||||
row_output = '; '.join([f'{key}: {value}' for key, value in row.items()])
|
||||
# print(row_output)
|
||||
paragraphs.append({'title': '', 'content': row_output})
|
||||
|
||||
result.append({'name': sheet.name, 'paragraphs': paragraphs})
|
||||
|
||||
except BaseException as e:
|
||||
max_kb.error(f'excel split handle error: {e}')
|
||||
return [{'name': file.name, 'paragraphs': []}]
|
||||
return result
|
||||
|
||||
def get_content(self, file, save_image):
|
||||
# 打开 .xls 文件
|
||||
try:
|
||||
workbook = xlrd.open_workbook(file_contents=file.read(), formatting_info=True)
|
||||
sheets = workbook.sheets()
|
||||
md_tables = ''
|
||||
for sheet in sheets:
|
||||
# 过滤空白的sheet
|
||||
if sheet.nrows == 0 or sheet.ncols == 0:
|
||||
continue
|
||||
|
||||
# 获取表头和内容
|
||||
headers = sheet.row_values(0)
|
||||
data = [sheet.row_values(row_idx) for row_idx in range(1, sheet.nrows)]
|
||||
|
||||
# 构建 Markdown 表格
|
||||
md_table = '| ' + ' | '.join(headers) + ' |\n'
|
||||
md_table += '| ' + ' | '.join(['---'] * len(headers)) + ' |\n'
|
||||
for row in data:
|
||||
# 将每个单元格中的内容替换换行符为 <br> 以保留原始格式
|
||||
md_table += '| ' + ' | '.join(
|
||||
[str(cell).replace('\n', '<br>') if cell else '' for cell in row]) + ' |\n'
|
||||
md_tables += md_table + '\n\n'
|
||||
|
||||
return md_tables
|
||||
except Exception as e:
|
||||
max_kb.error(f'excel split handle error: {e}')
|
||||
return f'error: {e}'
|
||||
@@ -0,0 +1,122 @@
|
||||
import io
|
||||
|
||||
from loguru import logger
|
||||
from openpyxl import load_workbook
|
||||
|
||||
from bisheng.api.services.handler.base_parse_table_handle import BaseParseTableHandle
|
||||
from bisheng.api.services.handler.impl.tools import xlsx_embed_cells_images
|
||||
|
||||
|
||||
# from common.handle.base_parse_table_handle import BaseParseTableHandle
|
||||
# from common.handle.impl.tools import xlsx_embed_cells_images
|
||||
|
||||
# logger = logging.getLogger("logger")
|
||||
|
||||
|
||||
class XlsxSplitHandle(BaseParseTableHandle):
|
||||
|
||||
def support(self, file, get_buffer):
|
||||
file_name: str = file.name.lower()
|
||||
if file_name.endswith('.xlsx'):
|
||||
return True
|
||||
return False
|
||||
|
||||
def fill_merged_cells(self, sheet, image_dict):
|
||||
data = []
|
||||
|
||||
# 获取第一行作为标题行
|
||||
headers = []
|
||||
for idx, cell in enumerate(sheet[1]):
|
||||
if cell.value is None:
|
||||
headers.append(' ' * (idx + 1))
|
||||
else:
|
||||
headers.append(cell.value)
|
||||
|
||||
# 从第二行开始遍历每一行
|
||||
for row in sheet.iter_rows(min_row=2, values_only=False):
|
||||
row_data = {}
|
||||
for col_idx, cell in enumerate(row):
|
||||
cell_value = cell.value
|
||||
|
||||
# 如果单元格为空,并且该单元格在合并单元格内,获取合并单元格的值
|
||||
if cell_value is None:
|
||||
for merged_range in sheet.merged_cells.ranges:
|
||||
if cell.coordinate in merged_range:
|
||||
cell_value = sheet[merged_range.min_row][merged_range.min_col -
|
||||
1].value
|
||||
break
|
||||
|
||||
image = image_dict.get(cell_value, None)
|
||||
if image is not None:
|
||||
cell_value = f''
|
||||
|
||||
# 使用标题作为键,单元格的值作为值存入字典
|
||||
row_data[headers[col_idx]] = cell_value
|
||||
data.append(row_data)
|
||||
|
||||
return data
|
||||
|
||||
def handle(self, file, get_buffer, save_image):
|
||||
buffer = get_buffer(file)
|
||||
try:
|
||||
wb = load_workbook(io.BytesIO(buffer))
|
||||
try:
|
||||
image_dict: dict = xlsx_embed_cells_images(io.BytesIO(buffer))
|
||||
save_image([item for item in image_dict.values()])
|
||||
except Exception:
|
||||
image_dict = {}
|
||||
result = []
|
||||
for sheetname in wb.sheetnames:
|
||||
paragraphs = []
|
||||
ws = wb[sheetname]
|
||||
data = self.fill_merged_cells(ws, image_dict)
|
||||
|
||||
for row in data:
|
||||
row_output = '; '.join([f'{key}: {value}' for key, value in row.items()])
|
||||
# print(row_output)
|
||||
paragraphs.append({'title': '', 'content': row_output})
|
||||
|
||||
result.append({'name': sheetname, 'paragraphs': paragraphs})
|
||||
|
||||
except BaseException as e:
|
||||
logger.error(f'excel split handle error: {e}')
|
||||
return [{'name': file.name, 'paragraphs': []}]
|
||||
return result
|
||||
|
||||
def get_content(self, file, save_image):
|
||||
try:
|
||||
# 加载 Excel 文件
|
||||
workbook = load_workbook(file)
|
||||
try:
|
||||
image_dict: dict = xlsx_embed_cells_images(file)
|
||||
if len(image_dict) > 0:
|
||||
save_image(image_dict.values())
|
||||
except Exception as e:
|
||||
image_dict = {}
|
||||
md_tables = ''
|
||||
# 如果未指定 sheet_name,则使用第一个工作表
|
||||
for sheetname in workbook.sheetnames:
|
||||
sheet = workbook[sheetname] if sheetname else workbook.active
|
||||
rows = self.fill_merged_cells(sheet, image_dict)
|
||||
if len(rows) == 0:
|
||||
continue
|
||||
# 提取表头和内容
|
||||
|
||||
headers = [f'{key}' for key, value in rows[0].items()]
|
||||
|
||||
# 构建 Markdown 表格
|
||||
md_table = '| ' + ' | '.join(headers) + ' |\n'
|
||||
md_table += '| ' + ' | '.join(['---'] * len(headers)) + ' |\n'
|
||||
for row in rows:
|
||||
r = [f'{value}' for key, value in row.items()]
|
||||
md_table += '| ' + ' | '.join([
|
||||
str(cell).replace('\n', '<br>') if cell is not None else '' for cell in r
|
||||
]) + ' |\n'
|
||||
|
||||
md_tables += md_table + '\n\n'
|
||||
|
||||
md_tables = md_tables.replace('/api/image/', '/api/file/')
|
||||
return md_tables
|
||||
except Exception as e:
|
||||
logger.error(f'excel split handle error: {e}')
|
||||
return f'error: {e}'
|
||||
@@ -0,0 +1,116 @@
|
||||
"""
|
||||
@project: MaxKB
|
||||
@Author:虎
|
||||
@file: tools.py
|
||||
@date:2024/9/11 16:41
|
||||
@desc:
|
||||
"""
|
||||
import io
|
||||
from functools import reduce
|
||||
from io import BytesIO
|
||||
from xml.etree.ElementTree import fromstring
|
||||
from zipfile import ZipFile
|
||||
|
||||
from PIL import Image as PILImage
|
||||
from openpyxl.drawing.image import Image as openpyxl_Image
|
||||
from openpyxl.packaging.relationship import get_dependents, get_rels_path
|
||||
from openpyxl.xml.constants import REL_NS, SHEET_DRAWING_NS, SHEET_MAIN_NS
|
||||
|
||||
|
||||
# from common.handle.base_parse_qa_handle import get_title_row_index_dict, get_row_value
|
||||
# from dataset.models import Image
|
||||
|
||||
|
||||
def parse_element(element) -> {}:
|
||||
data = {}
|
||||
xdr_namespace = '{%s}' % SHEET_DRAWING_NS
|
||||
targets = level_order_traversal(element, xdr_namespace + 'nvPicPr')
|
||||
for target in targets:
|
||||
cNvPr = embed = ''
|
||||
for child in target:
|
||||
if child.tag == xdr_namespace + 'nvPicPr':
|
||||
cNvPr = child[0].attrib['name']
|
||||
elif child.tag == xdr_namespace + 'blipFill':
|
||||
_rel_embed = '{%s}embed' % REL_NS
|
||||
embed = child[0].attrib[_rel_embed]
|
||||
if cNvPr:
|
||||
data[cNvPr] = embed
|
||||
return data
|
||||
|
||||
|
||||
def parse_element_sheet_xml(element) -> []:
|
||||
data = []
|
||||
xdr_namespace = '{%s}' % SHEET_MAIN_NS
|
||||
targets = level_order_traversal(element, xdr_namespace + 'f')
|
||||
for target in targets:
|
||||
for child in target:
|
||||
if child.tag == xdr_namespace + 'f':
|
||||
data.append(child.text)
|
||||
return data
|
||||
|
||||
|
||||
def level_order_traversal(root, flag: str) -> []:
|
||||
queue = [root]
|
||||
targets = []
|
||||
while queue:
|
||||
node = queue.pop(0)
|
||||
children = [child.tag for child in node]
|
||||
if flag in children:
|
||||
targets.append(node)
|
||||
continue
|
||||
for child in node:
|
||||
queue.append(child)
|
||||
return targets
|
||||
|
||||
|
||||
def handle_images(deps, archive: ZipFile) -> []:
|
||||
images = []
|
||||
if not PILImage: # Pillow not installed, drop images
|
||||
return images
|
||||
for dep in deps:
|
||||
try:
|
||||
image_io = archive.read(dep.target)
|
||||
image = openpyxl_Image(BytesIO(image_io))
|
||||
except Exception as e:
|
||||
continue
|
||||
image.embed = dep.id # 文件rId
|
||||
image.target = dep.target # 文件地址
|
||||
images.append(image)
|
||||
return images
|
||||
|
||||
|
||||
def xlsx_embed_cells_images(buffer) -> {}:
|
||||
archive = ZipFile(buffer)
|
||||
# 解析cellImage.xml文件
|
||||
deps = get_dependents(archive, get_rels_path('xl/cellimages.xml'))
|
||||
image_rel = handle_images(deps=deps, archive=archive)
|
||||
# 工作表及其中图片ID
|
||||
sheet_list = {}
|
||||
for item in archive.namelist():
|
||||
if not item.startswith('xl/worksheets/sheet'):
|
||||
continue
|
||||
key = item.split('/')[-1].split('.')[0].split('sheet')[-1]
|
||||
sheet_list[key] = parse_element_sheet_xml(fromstring(archive.read(item)))
|
||||
cell_images_xml = parse_element(fromstring(archive.read('xl/cellimages.xml')))
|
||||
cell_images_rel = {}
|
||||
for image in image_rel:
|
||||
cell_images_rel[image.embed] = image
|
||||
for cnv, embed in cell_images_xml.items():
|
||||
cell_images_xml[cnv] = cell_images_rel.get(embed)
|
||||
result = {}
|
||||
for key, img in cell_images_xml.items():
|
||||
image_excel_id_list = [
|
||||
_xl for _xl in reduce(lambda x, y: [*x, *y],
|
||||
[sheet for sheet_id, sheet in sheet_list.items()], [])
|
||||
if key in _xl
|
||||
]
|
||||
if len(image_excel_id_list) > 0:
|
||||
# image_excel_id = image_excel_id_list[-1]
|
||||
f = archive.open(img.target)
|
||||
img_byte = io.BytesIO()
|
||||
im = PILImage.open(f).convert('RGB')
|
||||
im.save(img_byte, format='JPEG')
|
||||
# image = Image(id=uuid.uuid1(), image=img_byte.getvalue(), image_name=img.path)
|
||||
# result['=' + image_excel_id] = image
|
||||
archive.close()
|
||||
return result
|
||||
@@ -0,0 +1,85 @@
|
||||
"""
|
||||
@project: maxkb
|
||||
@Author:虎
|
||||
@file: xls_parse_qa_handle.py
|
||||
@date:2024/5/21 14:59
|
||||
@desc:
|
||||
"""
|
||||
from typing import List
|
||||
|
||||
import xlrd
|
||||
from bisheng.api.services.handler.base_split_handle import BaseSplitHandle
|
||||
|
||||
# from common.handle.base_split_handle import BaseSplitHandle
|
||||
|
||||
|
||||
def post_cell(cell_value):
|
||||
return cell_value.replace('\n', '<br>').replace('|', '|')
|
||||
|
||||
|
||||
def row_to_md(row):
|
||||
return '| ' + ' | '.join([post_cell(str(cell)) if cell is not None else ''
|
||||
for cell in row]) + ' |\n'
|
||||
|
||||
|
||||
def handle_sheet(file_name, sheet, limit: int):
|
||||
rows = iter([sheet.row_values(i) for i in range(sheet.nrows)])
|
||||
paragraphs = []
|
||||
result = {'name': file_name, 'content': paragraphs}
|
||||
try:
|
||||
title_row_list = next(rows)
|
||||
title_md_content = row_to_md(title_row_list)
|
||||
title_md_content += '| ' + ' | '.join(
|
||||
['---' if cell is not None else '' for cell in title_row_list]) + ' |\n'
|
||||
except Exception:
|
||||
return result
|
||||
if len(title_row_list) == 0:
|
||||
return result
|
||||
result_item_content = ''
|
||||
for row in rows:
|
||||
next_md_content = row_to_md(row)
|
||||
next_md_content_len = len(next_md_content)
|
||||
result_item_content_len = len(result_item_content)
|
||||
if len(result_item_content) == 0:
|
||||
result_item_content += title_md_content
|
||||
result_item_content += next_md_content
|
||||
else:
|
||||
if result_item_content_len + next_md_content_len < limit:
|
||||
result_item_content += next_md_content
|
||||
else:
|
||||
paragraphs.append({'content': result_item_content, 'title': ''})
|
||||
result_item_content = title_md_content + next_md_content
|
||||
if len(result_item_content) > 0:
|
||||
paragraphs.append({'content': result_item_content, 'title': ''})
|
||||
return result
|
||||
|
||||
|
||||
class XlsSplitHandle(BaseSplitHandle):
|
||||
|
||||
def handle(self, file_name, pattern_list: List, with_filter: bool, limit: int, file_path,
|
||||
save_image):
|
||||
with open(file_path, 'rb') as f:
|
||||
buffer = f.read()
|
||||
try:
|
||||
workbook = xlrd.open_workbook(file_contents=buffer)
|
||||
worksheets = workbook.sheets()
|
||||
worksheets_size = len(worksheets)
|
||||
return [
|
||||
row for row in [
|
||||
handle_sheet(file_name, sheet, limit) if worksheets_size == 1
|
||||
and sheet.name == 'Sheet1' else handle_sheet(sheet.name, sheet, limit)
|
||||
for sheet in worksheets
|
||||
] if row is not None
|
||||
]
|
||||
except Exception:
|
||||
return [{'name': file_name, 'content': []}]
|
||||
|
||||
def get_content(self, file, save_image):
|
||||
pass
|
||||
|
||||
def support(self, file_name: str, file_path: str):
|
||||
with open(file_path, 'rb') as f:
|
||||
buffer = f.read()
|
||||
if file_name.endswith('.xls') and xlrd.inspect_format(content=buffer):
|
||||
return True
|
||||
return False
|
||||
@@ -0,0 +1,97 @@
|
||||
"""
|
||||
@project: maxkb
|
||||
@Author:虎
|
||||
@file: xlsx_parse_qa_handle.py
|
||||
@date:2024/5/21 14:59
|
||||
@desc:
|
||||
"""
|
||||
import io
|
||||
from typing import List
|
||||
|
||||
import openpyxl
|
||||
from bisheng.api.services.handler.base_split_handle import BaseSplitHandle
|
||||
from bisheng.api.services.handler.impl.tools import xlsx_embed_cells_images
|
||||
|
||||
# from common.handle.base_split_handle import BaseSplitHandle
|
||||
# from common.handle.impl.tools import xlsx_embed_cells_images
|
||||
|
||||
|
||||
def post_cell(image_dict, cell_value):
|
||||
image = image_dict.get(cell_value, None)
|
||||
if image is not None:
|
||||
return f''
|
||||
return cell_value.replace('\n', '<br>').replace('|', '|')
|
||||
|
||||
|
||||
def row_to_md(row, image_dict):
|
||||
return '| ' + ' | '.join([
|
||||
post_cell(image_dict, str(cell.value if cell.value is not None else ''))
|
||||
if cell is not None else '' for cell in row
|
||||
]) + ' |\n'
|
||||
|
||||
|
||||
def handle_sheet(file_name, sheet, image_dict, limit: int):
|
||||
rows = sheet.rows
|
||||
paragraphs = []
|
||||
result = {'name': file_name, 'content': paragraphs}
|
||||
try:
|
||||
|
||||
title_row_list = next(rows)
|
||||
title_md_content = row_to_md(title_row_list, image_dict)
|
||||
title_md_content += '| ' + ' | '.join(
|
||||
['---' if cell is not None else '' for cell in title_row_list]) + ' |\n'
|
||||
except Exception:
|
||||
return result
|
||||
if len(title_row_list) == 0:
|
||||
return result
|
||||
result_item_content = ''
|
||||
for row in rows:
|
||||
next_md_content = row_to_md(row, image_dict)
|
||||
next_md_content_len = len(next_md_content)
|
||||
result_item_content_len = len(result_item_content)
|
||||
if len(result_item_content) == 0:
|
||||
result_item_content += title_md_content
|
||||
result_item_content += next_md_content
|
||||
else:
|
||||
if result_item_content_len + next_md_content_len < limit:
|
||||
result_item_content += next_md_content
|
||||
else:
|
||||
paragraphs.append({'content': result_item_content, 'title': ''})
|
||||
result_item_content = title_md_content + next_md_content
|
||||
if len(result_item_content) > 0:
|
||||
paragraphs.append({'content': result_item_content, 'title': ''})
|
||||
return result
|
||||
|
||||
|
||||
class XlsxSplitHandle(BaseSplitHandle):
|
||||
|
||||
def handle(self, file_name, pattern_list: List, with_filter: bool, limit: int, file_path,
|
||||
save_image):
|
||||
with open(file_path, 'rb') as f:
|
||||
buffer = f.read()
|
||||
try:
|
||||
workbook = openpyxl.load_workbook(io.BytesIO(buffer))
|
||||
try:
|
||||
image_dict: dict = xlsx_embed_cells_images(io.BytesIO(buffer))
|
||||
save_image([item for item in image_dict.values()])
|
||||
except Exception:
|
||||
image_dict = {}
|
||||
worksheets = workbook.worksheets
|
||||
worksheets_size = len(worksheets)
|
||||
return [
|
||||
row for row in [
|
||||
handle_sheet(file_name, sheet, image_dict, limit
|
||||
) if worksheets_size == 1 and sheet.title == 'Sheet1' else
|
||||
handle_sheet(sheet.title, sheet, image_dict, limit) for sheet in worksheets
|
||||
] if row is not None
|
||||
]
|
||||
except Exception:
|
||||
return [{'name': file_name, 'content': []}]
|
||||
|
||||
def get_content(self, file, save_image):
|
||||
pass
|
||||
|
||||
def support(self, file_name: str, file_path: str):
|
||||
if file_name.endswith('.xlsx'):
|
||||
return True
|
||||
return False
|
||||
@@ -0,0 +1,52 @@
|
||||
import random
|
||||
import string
|
||||
|
||||
|
||||
class VoucherGenerator:
|
||||
def __init__(self, length=10):
|
||||
self.length = length
|
||||
# 排除相像的字母和数字: 'I', 'l', 'O', '0', '1'
|
||||
self.characters = ''.join(set(string.ascii_letters + string.digits) - set('IlOo01'))
|
||||
self.weights = [7, 9, 10, 5, 8, 4, 2, 1, 3] # 加权因子
|
||||
self.check_digits = ['1', '0', 'X', '9', '8', '7', '6', '5', '4', '3', '2'] # 校验码对应表
|
||||
|
||||
def generate_voucher(self):
|
||||
voucher_base = ''.join(random.choices(self.characters, k=self.length - 1))
|
||||
check_digit = self.calculate_check_digit(voucher_base)
|
||||
return voucher_base + check_digit
|
||||
|
||||
def calculate_check_digit(self, voucher_base):
|
||||
total = sum(self.weights[i] * (ord(char) - ord('A') if char.isalpha() else int(char)) for i, char in
|
||||
enumerate(voucher_base))
|
||||
remainder = total % 11
|
||||
return self.check_digits[remainder]
|
||||
|
||||
def validate_voucher(self, voucher):
|
||||
if len(voucher) != 10:
|
||||
return False, "Invalid voucher length"
|
||||
|
||||
voucher_base = voucher[:-1]
|
||||
provided_check_digit = voucher[-1]
|
||||
|
||||
calculated_check_digit = self.calculate_check_digit(voucher_base)
|
||||
|
||||
if provided_check_digit == calculated_check_digit:
|
||||
return True, "Valid voucher"
|
||||
else:
|
||||
return False, "Invalid voucher"
|
||||
|
||||
|
||||
# 示例用法
|
||||
if __name__ == "__main__":
|
||||
generator = VoucherGenerator()
|
||||
voucher_code = generator.generate_voucher() # 生成一个唯一的兑换码
|
||||
print(f"Generated voucher code: {voucher_code}")
|
||||
|
||||
# 验证兑换码
|
||||
is_valid, info = generator.validate_voucher(voucher_code)
|
||||
print(f"Is valid: {is_valid}, Info: {info}")
|
||||
|
||||
# 尝试验证一个无效的兑换码
|
||||
invalid_voucher_code = 'ABCDEFGHJK967'
|
||||
is_valid, info = generator.validate_voucher(invalid_voucher_code)
|
||||
print(f"Is valid: {is_valid}, Info: {info}")
|
||||
@@ -0,0 +1,112 @@
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.services.invite_code.code_validator import VoucherGenerator
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.database.models.invite_code import InviteCode, InviteCodeDao
|
||||
from bisheng.utils import generate_uuid
|
||||
|
||||
|
||||
class InviteCodeService:
|
||||
|
||||
@classmethod
|
||||
async def use_invite_code(cls, user_id: int) -> bool:
|
||||
"""
|
||||
使用邀请码
|
||||
:param user_id: 用户ID
|
||||
:return: 邀请码使用结果
|
||||
"""
|
||||
logger.debug(f"use_invite_code {user_id}")
|
||||
|
||||
codes = await InviteCodeDao.get_user_bind_code(user_id)
|
||||
for one in codes:
|
||||
flag = await InviteCodeDao.use_invite_code(user_id, one.code)
|
||||
if flag:
|
||||
logger.debug(f"use_invite_code {user_id}, {one.code} success")
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
@classmethod
|
||||
async def revoke_invite_code(cls, user_id: int) -> bool:
|
||||
"""
|
||||
撤销邀请码
|
||||
:param user_id: 用户ID
|
||||
:return: 邀请码撤销结果
|
||||
"""
|
||||
logger.debug(f"revoke_invite_code {user_id}")
|
||||
|
||||
codes = await InviteCodeDao.get_user_all_code(user_id)
|
||||
for one in codes:
|
||||
# 说明是崭新的邀请码,未被使用
|
||||
if one.used <= 0:
|
||||
continue
|
||||
flag = await InviteCodeDao.revoke_invite_code_used(user_id, one.code)
|
||||
if flag:
|
||||
logger.debug(f"revoke_invite_code {user_id}, {one.code} success")
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
@classmethod
|
||||
async def create_batch_invite_codes(cls, login_user: UserPayload, name: str, num: int, limit: int) -> list[str]:
|
||||
"""
|
||||
批量创建邀请码
|
||||
:param login_user: 操作用户信息
|
||||
:param name: 邀请码名称
|
||||
:param num: 邀请码数量
|
||||
:param limit: 每个邀请码的使用次数
|
||||
:return: 创建的邀请码列表
|
||||
"""
|
||||
generator = VoucherGenerator()
|
||||
code_list = []
|
||||
batch_id = generate_uuid()
|
||||
for i in range(num):
|
||||
code_list.append(InviteCode(
|
||||
code=generator.generate_voucher(),
|
||||
batch_id=batch_id,
|
||||
batch_name=name,
|
||||
limit=limit,
|
||||
created_id=login_user.user_id,
|
||||
))
|
||||
# 检查生成的邀请码是否重复
|
||||
unique_codes = []
|
||||
for code in code_list:
|
||||
if code.code in unique_codes:
|
||||
raise ValueError(f"Duplicate invite code found: {code.code}")
|
||||
unique_codes.append(code.code)
|
||||
|
||||
# 调用数据库操作来保存邀请码
|
||||
await InviteCodeDao.insert_invite_code(code_list)
|
||||
return unique_codes
|
||||
|
||||
@classmethod
|
||||
async def get_invite_code_num(cls, login_user: UserPayload) -> int:
|
||||
"""
|
||||
获取用户可用的邀请码的使用次数
|
||||
:param login_user: 操作用户信息
|
||||
:return: 邀请码使用次数
|
||||
"""
|
||||
nums = 0
|
||||
codes = await InviteCodeDao.get_user_bind_code(login_user.user_id)
|
||||
for one in codes:
|
||||
nums += one.limit - one.used
|
||||
return nums
|
||||
|
||||
@classmethod
|
||||
async def bind_invite_code(cls, login_user: UserPayload, code: str) -> (bool, str):
|
||||
"""
|
||||
绑定邀请码
|
||||
:param login_user: 操作用户信息
|
||||
:param code: 邀请码
|
||||
:return: 绑定结果
|
||||
"""
|
||||
generator = VoucherGenerator()
|
||||
flag, _ = generator.validate_voucher(code)
|
||||
if not flag:
|
||||
return False, "您输入的邀请码无效"
|
||||
codes = await InviteCodeDao.get_user_bind_code(login_user.user_id)
|
||||
if codes:
|
||||
return False, "已绑定其他邀请码"
|
||||
|
||||
flag = await InviteCodeDao.bind_invite_code(login_user.user_id, code)
|
||||
return flag, "邀请码绑定成功" if flag else "您输入的邀请码无效"
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,269 @@
|
||||
import os
|
||||
import shutil # For checking if the executable is in PATH
|
||||
import subprocess
|
||||
|
||||
from loguru import logger
|
||||
|
||||
|
||||
def get_libreoffice_path():
|
||||
"""
|
||||
Tries to find the LibreOffice executable.
|
||||
prerequisites:
|
||||
|
||||
1. install libreoffice
|
||||
2. linux:
|
||||
sudo apt-get install libreoffice
|
||||
sudo yum install libreoffice-headless
|
||||
3. macos:
|
||||
brew install libreoffice
|
||||
"""
|
||||
if shutil.which("soffice"):
|
||||
return "soffice"
|
||||
if shutil.which("libreoffice"):
|
||||
return "libreoffice"
|
||||
# Common Windows paths
|
||||
windows_paths = [
|
||||
r"C:\Program Files\LibreOffice\program\soffice.exe",
|
||||
r"C:\Program Files (x86)\LibreOffice\program\soffice.exe",
|
||||
]
|
||||
for path in windows_paths:
|
||||
if os.path.exists(path):
|
||||
return path
|
||||
return None
|
||||
|
||||
|
||||
def convert_doc_to_docx(input_doc_path, output_dir=None):
|
||||
"""
|
||||
Converts a .doc file to .docx using LibreOffice/soffice command line.
|
||||
|
||||
Args:
|
||||
input_doc_path (str): The absolute path to the input .doc file.
|
||||
output_dir (str, optional): The directory to save the converted .docx file.
|
||||
If None, saves in the same directory as the input file.
|
||||
libreoffice_exec (str, optional): The command name or full path of the
|
||||
LibreOffice executable (e.g., 'libreoffice',
|
||||
'soffice', or '/opt/libreoffice7.x/program/soffice').
|
||||
|
||||
Returns:
|
||||
str: The path to the converted .docx file if successful, None otherwise.
|
||||
"""
|
||||
if not os.path.isabs(input_doc_path):
|
||||
input_doc_path = os.path.abspath(input_doc_path)
|
||||
|
||||
if not input_doc_path.lower().endswith(".doc"):
|
||||
logger.debug(f"Error: Input file '{input_doc_path}' is not a .doc file.")
|
||||
return None
|
||||
|
||||
if not os.path.exists(input_doc_path):
|
||||
logger.debug(f"Error: Input file not found at '{input_doc_path}'")
|
||||
return None
|
||||
|
||||
# Determine output directory
|
||||
if output_dir is None:
|
||||
output_dir = os.path.dirname(input_doc_path)
|
||||
else:
|
||||
if not os.path.isabs(output_dir):
|
||||
output_dir = os.path.abspath(output_dir)
|
||||
if not os.path.exists(output_dir):
|
||||
try:
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
logger.debug(f"Created output directory: '{output_dir}'")
|
||||
except OSError as e:
|
||||
logger.debug(f"Error creating output directory '{output_dir}': {e}")
|
||||
return None
|
||||
|
||||
# Check if libreoffice_exec is in PATH if it's not a full path
|
||||
soffice_path = get_libreoffice_path()
|
||||
if not soffice_path:
|
||||
logger.debug(
|
||||
"Error: LibreOffice (soffice) command not found. Please install LibreOffice and ensure it's in your PATH, or adjust 'get_libreoffice_path()'."
|
||||
)
|
||||
return False
|
||||
|
||||
# Construct the output .docx file path
|
||||
base_name = os.path.basename(input_doc_path)
|
||||
file_name_no_ext = os.path.splitext(base_name)[0]
|
||||
output_docx_path = os.path.join(output_dir, f"{file_name_no_ext}.docx")
|
||||
|
||||
command = [
|
||||
soffice_path,
|
||||
"--headless", # Run in headless mode (no GUI)
|
||||
"--convert-to",
|
||||
"docx", # Specify the output format
|
||||
"--outdir",
|
||||
output_dir, # Specify the output directory
|
||||
input_doc_path, # The input file
|
||||
]
|
||||
|
||||
logger.debug(f"Executing command: {' '.join(command)}")
|
||||
|
||||
try:
|
||||
process = subprocess.run(
|
||||
command, check=True, capture_output=True, text=True, timeout=120
|
||||
) # 120 seconds timeout
|
||||
logger.debug(f"LibreOffice STDOUT: {process.stdout}")
|
||||
if (
|
||||
process.stderr
|
||||
): # LibreOffice sometimes logger.debugs info to stderr even on success
|
||||
logger.debug(f"LibreOffice STDERR: {process.stderr}")
|
||||
|
||||
# Check if the file was actually created
|
||||
# LibreOffice creates the file with the correct name in the output_dir
|
||||
expected_file_in_outdir = os.path.join(output_dir, f"{file_name_no_ext}.docx")
|
||||
if os.path.exists(expected_file_in_outdir):
|
||||
# If output_docx_path is different (it shouldn't be with this logic, but for safety)
|
||||
if expected_file_in_outdir != output_docx_path:
|
||||
shutil.move(expected_file_in_outdir, output_docx_path)
|
||||
logger.debug(
|
||||
f"Successfully converted '{input_doc_path}' to '{output_docx_path}'"
|
||||
)
|
||||
return output_docx_path
|
||||
else:
|
||||
# This case should ideally not happen if subprocess.run didn't raise an error
|
||||
# and LibreOffice worked as expected.
|
||||
logger.debug(
|
||||
f"Error: Conversion command seemed to succeed, but output file '{expected_file_in_outdir}' not found."
|
||||
)
|
||||
logger.debug(
|
||||
"Please check LibreOffice's behavior and output directory permissions."
|
||||
)
|
||||
return None
|
||||
|
||||
except FileNotFoundError:
|
||||
logger.debug(
|
||||
f"Error: The LibreOffice executable '{soffice_path}' was not found."
|
||||
)
|
||||
logger.debug(
|
||||
"Ensure LibreOffice is installed and the command is in your PATH or provide the full path."
|
||||
)
|
||||
return None
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.debug(f"Error during LibreOffice conversion for '{input_doc_path}':")
|
||||
logger.debug(f"Command: {' '.join(e.cmd)}")
|
||||
logger.debug(f"Return code: {e.returncode}")
|
||||
logger.debug(f"STDOUT: {e.stdout}")
|
||||
logger.debug(f"STDERR: {e.stderr}")
|
||||
return None
|
||||
except subprocess.TimeoutExpired:
|
||||
logger.debug(f"Error: LibreOffice conversion for '{input_doc_path}' timed out.")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.debug(
|
||||
f"An unexpected error occurred during conversion of '{input_doc_path}': {e}"
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def convert_ppt_to_pdf(input_path, output_dir=None):
|
||||
"""
|
||||
Converts .ppt or .pptx to PDF using LibreOffice soffice command.
|
||||
|
||||
Args:
|
||||
input_path (str): Path to the .ppt or .pptx file.
|
||||
output_dir (str, optional): Directory to save the PDF.
|
||||
Defaults to the same directory as the input file.
|
||||
"""
|
||||
if not (
|
||||
input_path.lower().endswith(".ppt") or input_path.lower().endswith(".pptx")
|
||||
):
|
||||
logger.debug(f"Error: {input_path} is not a .ppt or .pptx file.")
|
||||
return False
|
||||
|
||||
if not os.path.exists(input_path):
|
||||
logger.debug(f"Error: File not found at {input_path}")
|
||||
return False
|
||||
|
||||
soffice_path = get_libreoffice_path()
|
||||
if not soffice_path:
|
||||
logger.debug(
|
||||
"Error: LibreOffice (soffice) command not found. Please install LibreOffice and ensure it's in your PATH, or adjust 'get_libreoffice_path()'."
|
||||
)
|
||||
return False
|
||||
|
||||
if not output_dir:
|
||||
output_dir = os.path.dirname(input_path)
|
||||
else:
|
||||
if not os.path.exists(output_dir):
|
||||
os.makedirs(output_dir)
|
||||
|
||||
# The output PDF will have the same name as the input file, but with a .pdf extension,
|
||||
# and will be placed in the output_dir.
|
||||
base_name = os.path.basename(input_path)
|
||||
pdf_name = os.path.splitext(base_name)[0] + ".pdf"
|
||||
expected_pdf_path = os.path.join(output_dir, pdf_name)
|
||||
|
||||
command = [
|
||||
soffice_path,
|
||||
"--headless",
|
||||
"--convert-to",
|
||||
"pdf",
|
||||
"--outdir",
|
||||
output_dir,
|
||||
input_path,
|
||||
]
|
||||
|
||||
try:
|
||||
logger.debug(f"Converting {input_path} to PDF using {soffice_path}...")
|
||||
# LibreOffice can sometimes be slow to start up and convert.
|
||||
# It may also not provide much stdout/stderr unless there's a significant error.
|
||||
process = subprocess.run(
|
||||
command, capture_output=True, text=True, check=True, timeout=180
|
||||
) # 180 seconds timeout
|
||||
|
||||
if process.stdout:
|
||||
logger.debug(f"soffice stdout: {process.stdout}") # Often empty on success
|
||||
if process.stderr:
|
||||
logger.debug(
|
||||
f"soffice stderr: {process.stderr}"
|
||||
) # Check for any warnings/errors
|
||||
|
||||
if os.path.exists(expected_pdf_path):
|
||||
logger.debug(f"Successfully converted {input_path} to {expected_pdf_path}")
|
||||
return expected_pdf_path
|
||||
else:
|
||||
logger.debug(
|
||||
f"Conversion command ran, but output PDF not found at expected location: {expected_pdf_path}"
|
||||
)
|
||||
logger.debug(
|
||||
"Please check LibreOffice's behavior. Stdout/Stderr from above might provide clues."
|
||||
)
|
||||
return False
|
||||
|
||||
except (
|
||||
FileNotFoundError
|
||||
): # Should be caught by get_libreoffice_path, but as a fallback
|
||||
logger.debug(
|
||||
f"Error: {soffice_path} command not found. Please install LibreOffice and ensure it's in your PATH."
|
||||
)
|
||||
return False
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.debug(f"Error during soffice conversion for {input_path}: {e}")
|
||||
logger.debug(f"Exit code: {e.returncode}")
|
||||
logger.debug(f"Stdout: {e.stdout}")
|
||||
logger.debug(f"Stderr: {e.stderr}")
|
||||
# LibreOffice might return a non-zero exit code even for some warnings.
|
||||
# Check if the file was created anyway.
|
||||
if os.path.exists(expected_pdf_path):
|
||||
logger.debug(
|
||||
f"Warning: soffice returned an error code, but PDF was created at {expected_pdf_path}"
|
||||
)
|
||||
return expected_pdf_path
|
||||
return False
|
||||
except subprocess.TimeoutExpired:
|
||||
logger.debug(f"Error: soffice conversion for {input_path} timed out.")
|
||||
# Check if the file was partially created or created despite timeout
|
||||
if os.path.exists(expected_pdf_path):
|
||||
logger.debug(
|
||||
f"Warning: soffice timed out, but PDF might have been created at {expected_pdf_path}"
|
||||
)
|
||||
return expected_pdf_path
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.debug(f"An unexpected error occurred with soffice for {input_path}: {e}")
|
||||
return False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
file_name = "/Users/tju/Resources/docs/docx/resume.doc"
|
||||
convert_doc_to_docx(file_name, output_dir="/Users/tju/Resources/docs/docx")
|
||||
logger.debug(f"{os.path.basename(file_name)}/x")
|
||||
@@ -0,0 +1,73 @@
|
||||
from starlette.websockets import WebSocket
|
||||
|
||||
from bisheng.linsight.state_message_manager import LinsightStateMessageManager, MessageData, MessageEventType
|
||||
|
||||
|
||||
class MessageStreamHandle(object):
|
||||
def __init__(self, websocket: 'WebSocket', session_version_id: str):
|
||||
"""
|
||||
初始化 MessageStreamHandle
|
||||
:param websocket:
|
||||
"""
|
||||
self._websocket = websocket
|
||||
self.session_version_id = session_version_id
|
||||
self._state_message_manager: LinsightStateMessageManager = LinsightStateMessageManager(
|
||||
session_version_id=session_version_id)
|
||||
|
||||
async def send_message(self, message_data: str) -> None:
|
||||
"""
|
||||
发送消息到 WebSocket
|
||||
:param message_data: 要发送的消息内容
|
||||
"""
|
||||
await self._websocket.send_text(message_data)
|
||||
|
||||
async def receive_message(self) -> str:
|
||||
"""
|
||||
接收来自 WebSocket 的消息
|
||||
:return:
|
||||
"""
|
||||
return await self._websocket.receive_text()
|
||||
|
||||
async def send_json(self, json_data: dict) -> None:
|
||||
"""
|
||||
发送 JSON 数据到 WebSocket
|
||||
:param json_data: 要发送的 JSON 数据
|
||||
"""
|
||||
await self._websocket.send_json(json_data)
|
||||
|
||||
async def receive_json(self) -> dict:
|
||||
"""
|
||||
接收来自 WebSocket 的 JSON 数据
|
||||
:return:
|
||||
"""
|
||||
return await self._websocket.receive_json()
|
||||
|
||||
# 处理 WebSocket 连接的生命周期事件
|
||||
async def connect(self) -> None:
|
||||
"""
|
||||
连接到 WebSocket
|
||||
"""
|
||||
await self._websocket.accept()
|
||||
|
||||
while True:
|
||||
try:
|
||||
message = await self._state_message_manager.pop_message()
|
||||
if message:
|
||||
await self.send_json(message.model_dump())
|
||||
|
||||
if message.event_type in [MessageEventType.ERROR_MESSAGE, MessageEventType.TASK_TERMINATED,
|
||||
MessageEventType.FINAL_RESULT]:
|
||||
await self._websocket.close(code=1000, reason="Session finished or error occurred")
|
||||
break
|
||||
|
||||
except Exception as e:
|
||||
await self.send_json(
|
||||
MessageData(event_type=MessageEventType.ERROR_MESSAGE, data={"error": str(e)}).model_dump())
|
||||
await self._websocket.close(code=1000, reason=f"Error: {str(e)}")
|
||||
break
|
||||
|
||||
async def disconnect(self) -> None:
|
||||
"""
|
||||
断开 WebSocket 连接
|
||||
"""
|
||||
await self._websocket.close(code=1000, reason="Client disconnected")
|
||||
@@ -0,0 +1,493 @@
|
||||
import json
|
||||
import uuid
|
||||
from typing import List, Dict
|
||||
|
||||
from langchain_core.documents import Document
|
||||
from langchain_core.embeddings import Embeddings
|
||||
from langchain_text_splitters import RecursiveCharacterTextSplitter
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.errcode.base import NotFoundError, ServerError
|
||||
from bisheng.api.services.knowledge_imp import decide_vectorstores
|
||||
from bisheng.api.services.llm import LLMService
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.v1.schema.inspiration_schema import SOPManagementSchema, SOPManagementUpdateSchema
|
||||
from bisheng.api.v1.schema.linsight_schema import SopRecordRead
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel, resp_200
|
||||
from bisheng.core.app_context import app_ctx
|
||||
from bisheng.database.models.linsight_sop import LinsightSOP, LinsightSOPDao, LinsightSOPRecord
|
||||
from bisheng.database.models.llm_server import LLMDao, LLMModelType
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.interface.embeddings.custom import FakeEmbedding
|
||||
from bisheng.interface.llms.custom import BishengLLM
|
||||
from bisheng.utils import util
|
||||
from bisheng.utils.embedding import decide_embeddings
|
||||
from bisheng_langchain.rag.init_retrievers import KeywordRetriever, BaselineVectorRetriever
|
||||
from bisheng_langchain.retrievers import EnsembleRetriever
|
||||
from bisheng_langchain.vectorstores import ElasticKeywordsSearch, Milvus
|
||||
|
||||
|
||||
class SOPManageService:
|
||||
__doc__ = "灵思SOP管理服务"
|
||||
|
||||
collection_name = "col_linsight_sop"
|
||||
|
||||
@staticmethod
|
||||
async def generate_sop_summary(sop_content: str, llm: BishengLLM = None) -> Dict[str, str]:
|
||||
"""生成SOP摘要"""
|
||||
default_summary = {"sop_title": "SOP名称", "sop_description": "SOP描述"}
|
||||
|
||||
try:
|
||||
if llm is None:
|
||||
workbench_conf = await LLMService.get_workbench_llm()
|
||||
llm = BishengLLM(model_id=workbench_conf.task_model.id, temperature=0)
|
||||
prompt_service = app_ctx.get_prompt_loader()
|
||||
prompt_obj = prompt_service.render_prompt(
|
||||
namespace="sop",
|
||||
prompt_name="gen_sop_summary",
|
||||
sop_detail=sop_content
|
||||
)
|
||||
|
||||
prompt = [
|
||||
("system", prompt_obj.prompt.system),
|
||||
("user", prompt_obj.prompt.user)
|
||||
]
|
||||
|
||||
response = await llm.ainvoke(prompt)
|
||||
if not response.content:
|
||||
return default_summary
|
||||
|
||||
return json.loads(response.content)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"生成SOP摘要失败: {e}")
|
||||
return default_summary
|
||||
|
||||
@staticmethod
|
||||
async def add_sop_record(sop_record: LinsightSOPRecord) -> LinsightSOPRecord:
|
||||
"""
|
||||
添加SOP记录
|
||||
"""
|
||||
if not sop_record.description:
|
||||
sop_summary = await SOPManageService.generate_sop_summary(sop_record.content, None)
|
||||
sop_record.description = sop_summary["sop_description"]
|
||||
|
||||
return await LinsightSOPDao.create_sop_record(sop_record)
|
||||
|
||||
@staticmethod
|
||||
async def get_sop_record(keyword: str = None, sort: str = None, page: int = 1, page_size: int = 10) -> \
|
||||
(List[SopRecordRead], int):
|
||||
"""
|
||||
根据关键词查询SOP记录
|
||||
"""
|
||||
user_ids = []
|
||||
if keyword:
|
||||
# 如果有关键词,先获取用户ID列表
|
||||
user_ids = await UserDao.afilter_users(user_ids=[], keyword=keyword)
|
||||
user_ids = [one.user_id for one in user_ids]
|
||||
|
||||
res = await LinsightSOPDao.filter_sop_record(keyword, user_ids, page, page_size, sort)
|
||||
count = await LinsightSOPDao.count_sop_record(keyword, user_ids)
|
||||
if not res:
|
||||
return [], 0
|
||||
|
||||
all_users = await UserDao.afilter_users(user_ids=[one.user_id for one in res])
|
||||
all_users = {
|
||||
one.user_id: one.user_name for one in all_users
|
||||
}
|
||||
|
||||
result = []
|
||||
for one in res:
|
||||
new_one = SopRecordRead.model_validate(one)
|
||||
new_one.user_name = all_users.get(one.user_id, str(one.user_id))
|
||||
result.append(new_one)
|
||||
return result, count
|
||||
|
||||
@staticmethod
|
||||
async def update_sop_record_score(session_version_id: str, score: int) -> None:
|
||||
await LinsightSOPDao.update_sop_record_score(session_version_id, score)
|
||||
|
||||
@staticmethod
|
||||
async def sync_sop_record(record_ids: list[int], override: bool = False, save_new: bool = False) \
|
||||
-> list[str] | None:
|
||||
|
||||
"""
|
||||
如果有重复的SOP记录,返回重复的记录名称列表
|
||||
"""
|
||||
sop_records = await LinsightSOPDao.get_sop_record_by_ids(record_ids)
|
||||
records_name_dict = {}
|
||||
repeat_names = set()
|
||||
name_set = set()
|
||||
sop_list = []
|
||||
oversize_records = []
|
||||
new_records = []
|
||||
for one in sop_records:
|
||||
if len(one.content) > 50000:
|
||||
oversize_records.append(one.name)
|
||||
continue
|
||||
new_records.append(one)
|
||||
if one.name not in name_set:
|
||||
records_name_dict[one.name] = one
|
||||
name_set.add(one.name)
|
||||
sop_records = new_records
|
||||
if not sop_records and oversize_records:
|
||||
raise ValueError(f"{'、'.join(oversize_records)}内容超长")
|
||||
if name_set:
|
||||
sop_list = await LinsightSOPDao.get_sops_by_names(list(name_set))
|
||||
for one in sop_list:
|
||||
repeat_names.add(one.name)
|
||||
|
||||
if override:
|
||||
# 先更新已有的sop库
|
||||
override_name_dict = {}
|
||||
for one in sop_list:
|
||||
if one_record := records_name_dict.get(one.name):
|
||||
await SOPManageService.update_sop(SOPManagementUpdateSchema(
|
||||
id=one.id,
|
||||
name=one.name,
|
||||
description=one_record.description,
|
||||
content=one_record.content,
|
||||
rating=one_record.rating,
|
||||
))
|
||||
override_name_dict[one.name] = True
|
||||
# 再新增剩下的sop记录
|
||||
for one in records_name_dict.values():
|
||||
if one not in override_name_dict:
|
||||
continue
|
||||
await SOPManageService.add_sop(SOPManagementSchema(
|
||||
name=one.name,
|
||||
description=one.description,
|
||||
content=one.content,
|
||||
rating=one.rating,
|
||||
), one.user_id)
|
||||
elif save_new:
|
||||
for one in sop_records:
|
||||
new_name = one.name
|
||||
if new_name in repeat_names:
|
||||
# 如果有重复的记录,添加后缀, 长度限制500个字符
|
||||
new_name = f"{one.name}副本"
|
||||
await SOPManageService.add_sop(SOPManagementSchema(
|
||||
name=new_name,
|
||||
description=one.description,
|
||||
content=one.content,
|
||||
rating=one.rating,
|
||||
), one.user_id)
|
||||
else:
|
||||
# 说明有重复的记录,需要用户确认
|
||||
if sop_list:
|
||||
return list(repeat_names)
|
||||
# 将记录插入到数据库中
|
||||
for one in sop_records:
|
||||
await SOPManageService.add_sop(SOPManagementSchema(
|
||||
name=one.name,
|
||||
description=one.description,
|
||||
content=one.content,
|
||||
rating=one.rating,
|
||||
), one.user_id)
|
||||
if oversize_records:
|
||||
raise ValueError(f"{'、'.join(oversize_records)}内容超长")
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
async def add_sop(sop_obj: SOPManagementSchema, user_id) -> UnifiedResponseModel | None:
|
||||
"""
|
||||
添加新的SOP
|
||||
:param user_id:
|
||||
:param sop_obj:
|
||||
:return: 添加的SOP对象
|
||||
"""
|
||||
|
||||
# 获取当前全局配置的embedding模型
|
||||
workbench_conf = await LLMService.get_workbench_llm()
|
||||
try:
|
||||
emb_model_id = workbench_conf.embedding_model.id
|
||||
if not emb_model_id:
|
||||
raise ServerError.http_exception(msg="未配置知识库embedding模型,请从工作台配置中设置")
|
||||
except AttributeError:
|
||||
raise ServerError.http_exception(msg="工作台配置中未找到SOP embedding模型,请从工作台配置中设置")
|
||||
|
||||
# 校验embedding模型
|
||||
embed_info = LLMDao.get_model_by_id(int(emb_model_id))
|
||||
if not embed_info:
|
||||
raise ServerError.http_exception(msg="知识库embedding模型不存在,请从工作台配置中设置")
|
||||
if embed_info.model_type != LLMModelType.EMBEDDING.value:
|
||||
raise ValueError("知识库embedding模型类型错误,请从工作台配置中设置")
|
||||
|
||||
vector_store_id = uuid.uuid4().hex
|
||||
|
||||
embeddings = decide_embeddings(emb_model_id)
|
||||
try:
|
||||
vector_client: Milvus = decide_vectorstores(
|
||||
SOPManageService.collection_name, "Milvus", embeddings
|
||||
)
|
||||
|
||||
es_client: ElasticKeywordsSearch = decide_vectorstores(
|
||||
SOPManageService.collection_name, "ElasticKeywordsSearch", FakeEmbedding()
|
||||
)
|
||||
metadatas = [{"vector_store_id": vector_store_id}]
|
||||
vector_client.add_texts([sop_obj.content[0:10000]], metadatas=metadatas)
|
||||
es_client.add_texts([sop_obj.content], ids=[vector_store_id], metadatas=metadatas)
|
||||
except Exception as e:
|
||||
raise ServerError.http_exception(msg=f"添加SOP失败,向向量存储添加数据失败: {str(e)}")
|
||||
|
||||
sop_dict = sop_obj.model_dump(exclude_unset=True)
|
||||
sop_dict["vector_store_id"] = vector_store_id # 设置向量存储ID
|
||||
# 这里可以添加数据库操作,将sop_obj保存到数据库中
|
||||
sop_model = LinsightSOP(**sop_dict)
|
||||
sop_model.user_id = user_id
|
||||
sop_model = await LinsightSOPDao.create_sop(sop_model)
|
||||
if not sop_model:
|
||||
raise ServerError.http_exception(msg="添加SOP失败")
|
||||
|
||||
return resp_200(data=sop_model)
|
||||
|
||||
@staticmethod
|
||||
async def update_sop(sop_obj: SOPManagementUpdateSchema) -> UnifiedResponseModel | None:
|
||||
"""
|
||||
更新SOP
|
||||
:param sop_obj:
|
||||
:return: 更新后的SOP对象
|
||||
"""
|
||||
# 校验SOP是否存在
|
||||
existing_sop = await LinsightSOPDao.get_sops_by_ids([sop_obj.id])
|
||||
if not existing_sop:
|
||||
raise NotFoundError.http_exception(msg="SOP不存在")
|
||||
|
||||
if sop_obj.content != existing_sop[0].content:
|
||||
|
||||
# 获取当前全局配置的embedding模型
|
||||
workbench_conf = await LLMService.get_workbench_llm()
|
||||
try:
|
||||
emb_model_id = workbench_conf.embedding_model.id
|
||||
if not emb_model_id:
|
||||
raise ServerError.http_exception(msg="未配置知识库embedding模型,请从工作台配置中设置")
|
||||
except AttributeError:
|
||||
raise ServerError.http_exception(msg="工作台配置中未找到SOP embedding模型,请从工作台配置中设置")
|
||||
|
||||
vector_store_id = existing_sop[0].vector_store_id
|
||||
embeddings = decide_embeddings(emb_model_id)
|
||||
|
||||
# 更新向量存储
|
||||
try:
|
||||
vector_client: Milvus = decide_vectorstores(
|
||||
SOPManageService.collection_name, "Milvus", embeddings
|
||||
)
|
||||
es_client: ElasticKeywordsSearch = decide_vectorstores(
|
||||
SOPManageService.collection_name, "ElasticKeywordsSearch", FakeEmbedding()
|
||||
)
|
||||
|
||||
vector_client.delete(expr=f"vector_store_id == '{vector_store_id}'")
|
||||
es_client.delete([vector_store_id])
|
||||
metadatas = [{"vector_store_id": vector_store_id}]
|
||||
vector_client.add_texts([sop_obj.content[0:10000]], metadatas=metadatas)
|
||||
es_client.add_texts([sop_obj.content], ids=[vector_store_id], metadatas=metadatas)
|
||||
|
||||
except Exception as e:
|
||||
raise ServerError.http_exception(msg=f"更新SOP失败,向向量存储更新数据失败: {str(e)}")
|
||||
|
||||
# 更新数据库中的SOP
|
||||
sop_model = await LinsightSOPDao.update_sop(sop_obj)
|
||||
|
||||
return resp_200(data=sop_model)
|
||||
|
||||
@staticmethod
|
||||
async def remove_sop(sop_ids: list[int], login_user: UserPayload) -> UnifiedResponseModel | None:
|
||||
"""
|
||||
删除SOP
|
||||
:param login_user:
|
||||
:param sop_ids: SOP唯一ID列表
|
||||
:return: 删除结果
|
||||
"""
|
||||
if not sop_ids:
|
||||
raise NotFoundError.http_exception(msg="SOP ID列表不能为空")
|
||||
|
||||
# 校验SOP是否存在
|
||||
existing_sops = await LinsightSOPDao.get_sops_by_ids(sop_ids)
|
||||
if not existing_sops:
|
||||
return resp_200(data=True)
|
||||
|
||||
# 删除向量存储中的数据
|
||||
try:
|
||||
vector_store_ids = [sop.vector_store_id for sop in existing_sops]
|
||||
vector_client: Milvus = decide_vectorstores(
|
||||
SOPManageService.collection_name, "Milvus", FakeEmbedding()
|
||||
)
|
||||
es_client: ElasticKeywordsSearch = decide_vectorstores(
|
||||
SOPManageService.collection_name, "ElasticKeywordsSearch", FakeEmbedding()
|
||||
)
|
||||
|
||||
vector_client.delete(expr=f"vector_store_id in {vector_store_ids}")
|
||||
es_client.delete(vector_store_ids)
|
||||
|
||||
except Exception as e:
|
||||
raise ServerError.http_exception(msg=f"删除SOP失败,向向量存储删除数据失败: {str(e)}")
|
||||
|
||||
# 删除数据库中的SOP
|
||||
await LinsightSOPDao.remove_sop(sop_ids=sop_ids)
|
||||
|
||||
return resp_200(data=True)
|
||||
|
||||
# sop 库检索
|
||||
@classmethod
|
||||
async def search_sop(cls, query: str, k: int = 3) -> (List[Document], str | None):
|
||||
"""
|
||||
搜索SOP
|
||||
:param k:
|
||||
:param query: 搜索关键词
|
||||
:return: 搜索结果
|
||||
"""
|
||||
# 获取当前全局配置的embedding模型
|
||||
try:
|
||||
vector_search = True
|
||||
es_search = True
|
||||
error_msg = None
|
||||
workbench_conf = await LLMService.get_workbench_llm()
|
||||
if workbench_conf.embedding_model is None or not workbench_conf.embedding_model.id:
|
||||
vector_search = False
|
||||
error_msg = "请联系管理员检查工作台向量检索模型状态"
|
||||
else:
|
||||
try:
|
||||
emb_model_id = workbench_conf.embedding_model.id
|
||||
embeddings = decide_embeddings(emb_model_id)
|
||||
await embeddings.aembed_query("test")
|
||||
except Exception as e:
|
||||
logger.error(f"向量检索模型初始化失败: {str(e)}")
|
||||
vector_search = False
|
||||
error_msg = "请联系管理员检查工作台向量检索模型状态"
|
||||
|
||||
# 创建文本分割器
|
||||
text_splitter = RecursiveCharacterTextSplitter()
|
||||
retrievers = []
|
||||
if vector_search and es_search:
|
||||
emb_model_id = workbench_conf.embedding_model.id
|
||||
embeddings = decide_embeddings(emb_model_id)
|
||||
|
||||
vector_client: Milvus = decide_vectorstores(
|
||||
SOPManageService.collection_name, "Milvus", embeddings
|
||||
)
|
||||
|
||||
es_client: ElasticKeywordsSearch = decide_vectorstores(
|
||||
SOPManageService.collection_name, "ElasticKeywordsSearch", FakeEmbedding()
|
||||
)
|
||||
|
||||
keyword_retriever = KeywordRetriever(keyword_store=es_client, search_kwargs={"k": 100},
|
||||
text_splitter=text_splitter)
|
||||
baseline_vector_retriever = BaselineVectorRetriever(vector_store=vector_client,
|
||||
search_kwargs={"k": 100},
|
||||
text_splitter=text_splitter)
|
||||
|
||||
retrievers = [keyword_retriever, baseline_vector_retriever]
|
||||
|
||||
elif es_search and not vector_search:
|
||||
# 仅使用关键词检索
|
||||
es_client: ElasticKeywordsSearch = decide_vectorstores(
|
||||
SOPManageService.collection_name, "ElasticKeywordsSearch", FakeEmbedding()
|
||||
)
|
||||
keyword_retriever = KeywordRetriever(keyword_store=es_client, search_kwargs={"k": 100},
|
||||
text_splitter=text_splitter)
|
||||
retrievers = [keyword_retriever]
|
||||
|
||||
elif vector_search and not es_search:
|
||||
# 仅使用向量检索
|
||||
emb_model_id = workbench_conf.embedding_model.id
|
||||
embeddings = decide_embeddings(emb_model_id)
|
||||
|
||||
vector_client: Milvus = decide_vectorstores(
|
||||
SOPManageService.collection_name, "Milvus", embeddings
|
||||
)
|
||||
|
||||
baseline_vector_retriever = BaselineVectorRetriever(vector_store=vector_client,
|
||||
search_kwargs={"k": 100},
|
||||
text_splitter=text_splitter)
|
||||
retrievers = [baseline_vector_retriever]
|
||||
else:
|
||||
error_msg = "SOP检索失败,向量检索与关键词检索均不可用"
|
||||
return [], error_msg
|
||||
|
||||
retriever = EnsembleRetriever(retrievers=retrievers, weights=[0.5, 0.5] if len(retrievers) > 1 else [1.0])
|
||||
|
||||
# 执行检索
|
||||
results = await retriever.ainvoke(input=query)
|
||||
|
||||
if not results:
|
||||
return [], error_msg
|
||||
|
||||
vector_store_ids = [doc.metadata.get("vector_store_id") for doc in results if
|
||||
doc.metadata.get("vector_store_id")]
|
||||
|
||||
# 根据vector_store_ids查询库中的sop
|
||||
sop_models = await LinsightSOPDao.get_sop_by_vector_store_ids(vector_store_ids)
|
||||
sop_model_vector_store_ids = [sop.vector_store_id for sop in sop_models]
|
||||
|
||||
# 过滤结果,确保只返回存在于数据库中的SOP
|
||||
results = [doc for doc in results if doc.metadata.get("vector_store_id") in sop_model_vector_store_ids]
|
||||
|
||||
# 过滤完取前k条结果
|
||||
results = results[:k]
|
||||
|
||||
return results, error_msg
|
||||
except Exception as e:
|
||||
logger.error(f"搜索SOP失败: {str(e)}")
|
||||
return [], f"SOP检索失败: {str(e)}"
|
||||
|
||||
# 重建SOP VectorStore
|
||||
@classmethod
|
||||
async def rebuild_sop_vector_store_task(cls, embeddings: Embeddings):
|
||||
"""
|
||||
重建SOP向量存储
|
||||
:return: 重建结果
|
||||
"""
|
||||
try:
|
||||
# 获取所有SOP
|
||||
all_sops = await LinsightSOPDao.get_all_sops()
|
||||
if not all_sops:
|
||||
logger.info("没有SOP数据需要重建向量存储")
|
||||
return None
|
||||
|
||||
# 包装同步函数为异步函数
|
||||
def sync_func(sops, emb):
|
||||
"""
|
||||
同步函数,用于重建SOP向量存储
|
||||
:param emb:
|
||||
:param sops:
|
||||
:return:
|
||||
"""
|
||||
|
||||
vector_client: Milvus = decide_vectorstores(
|
||||
SOPManageService.collection_name, "Milvus", emb
|
||||
)
|
||||
# 删除现有的向量存储collection
|
||||
if vector_client.col is not None:
|
||||
logger.info("删除现有的SOP向量存储collection")
|
||||
vector_client.col.drop()
|
||||
vector_client.col = None
|
||||
vector_client.fields = []
|
||||
|
||||
metadatas = [{"vector_store_id": sop.vector_store_id} for sop in sops]
|
||||
contents = [sop.content for sop in sops]
|
||||
|
||||
batch_size = 16
|
||||
for i in range(0, len(contents), batch_size):
|
||||
batch_contents = contents[i:i + batch_size]
|
||||
batch_metadatas = metadatas[i:i + batch_size]
|
||||
|
||||
# 添加新的SOP数据到向量存储
|
||||
vector_client.add_texts(batch_contents, metadatas=batch_metadatas)
|
||||
|
||||
logger.info("SOP向量存储重建完成: {}".format(len(sops)))
|
||||
|
||||
# 使用run_async运行同步函数
|
||||
await util.sync_func_to_async(sync_func)(all_sops, embeddings)
|
||||
return None
|
||||
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"重建SOP向量存储失败: {str(e)}")
|
||||
return None
|
||||
|
||||
# if __name__ == '__main__':
|
||||
# # 测试代码
|
||||
# results, error_msg = asyncio.run(SOPManageService.search_sop(query="投标文件编写指南", k=3))
|
||||
#
|
||||
# print(results)
|
||||
# print(error_msg)
|
||||
@@ -0,0 +1,949 @@
|
||||
import asyncio
|
||||
import os
|
||||
import uuid
|
||||
from dataclasses import dataclass
|
||||
from io import BytesIO
|
||||
from typing import Dict, List, Optional, AsyncGenerator, Tuple, Any
|
||||
from urllib.parse import unquote
|
||||
|
||||
from fastapi import UploadFile
|
||||
from langchain_core.tools import BaseTool
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.services.assistant_agent import AssistantAgent
|
||||
from bisheng.api.services.knowledge_imp import read_chunk_text, decide_vectorstores
|
||||
from bisheng.api.services.linsight.sop_manage import SOPManageService
|
||||
from bisheng.api.services.llm import LLMService
|
||||
from bisheng.api.services.tool import ToolServices
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.services.workstation import WorkStationService
|
||||
from bisheng.api.v1.schema.linsight_schema import LinsightQuestionSubmitSchema, BatchDownloadFilesSchema
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.cache.utils import save_file_to_folder, CACHE_DIR
|
||||
from bisheng.core.app_context import app_ctx
|
||||
from bisheng.database.models import LinsightSessionVersion
|
||||
from bisheng.database.models.flow import FlowType
|
||||
from bisheng.database.models.knowledge import KnowledgeRead, KnowledgeTypeEnum
|
||||
from bisheng.database.models.linsight_execute_task import LinsightExecuteTaskDao
|
||||
from bisheng.database.models.linsight_session_version import LinsightSessionVersionDao, SessionVersionStatusEnum
|
||||
from bisheng.database.models.linsight_sop import LinsightSOPRecord
|
||||
from bisheng.database.models.session import MessageSessionDao, MessageSession
|
||||
from bisheng.interface.embeddings.custom import FakeEmbedding
|
||||
from bisheng.interface.llms.custom import BishengLLM
|
||||
from bisheng.settings import settings
|
||||
from bisheng.utils import util
|
||||
from bisheng.utils.embedding import decide_embeddings
|
||||
from bisheng.utils.minio_client import minio_client
|
||||
from bisheng.utils.util import calculate_md5
|
||||
from bisheng_langchain.linsight.const import ExecConfig
|
||||
|
||||
|
||||
@dataclass
|
||||
class TaskNode:
|
||||
"""任务节点,用于构建任务树"""
|
||||
task: Any # LinsightExecuteTask 对象
|
||||
children: List['TaskNode'] = None
|
||||
|
||||
def __post_init__(self):
|
||||
if self.children is None:
|
||||
self.children = []
|
||||
|
||||
def to_dict(self) -> Dict:
|
||||
"""将任务节点转换为字典格式"""
|
||||
task_dict = self.task.model_dump()
|
||||
task_dict['children'] = [child.to_dict() for child in self.children]
|
||||
return task_dict
|
||||
|
||||
|
||||
class LinsightWorkbenchImpl:
|
||||
"""Linsight工作台实现类"""
|
||||
|
||||
# 类常量
|
||||
COLLECTION_NAME_PREFIX = "col_linsight_file_"
|
||||
FILE_INFO_REDIS_KEY_PREFIX = "linsight_file:"
|
||||
CACHE_EXPIRATION_HOURS = 24
|
||||
|
||||
class LinsightError(Exception):
|
||||
"""Linsight相关错误"""
|
||||
pass
|
||||
|
||||
class SearchSOPError(Exception):
|
||||
"""SOP检索错误"""
|
||||
|
||||
def __init__(self, message: str):
|
||||
super().__init__(message)
|
||||
self.message = message
|
||||
|
||||
class ToolsInitializationError(Exception):
|
||||
"""工具初始化错误"""
|
||||
pass
|
||||
|
||||
class BishengLLMError(Exception):
|
||||
"""Bisheng LLM相关错误"""
|
||||
pass
|
||||
|
||||
@classmethod
|
||||
async def submit_user_question(cls, submit_obj: LinsightQuestionSubmitSchema,
|
||||
login_user: UserPayload) -> tuple[MessageSession, LinsightSessionVersion]:
|
||||
"""
|
||||
提交用户问题并创建会话
|
||||
|
||||
Args:
|
||||
submit_obj: 提交的问题对象
|
||||
login_user: 登录用户信息
|
||||
|
||||
Returns:
|
||||
tuple: (消息会话模型, 灵思会话版本模型)
|
||||
|
||||
Raises:
|
||||
LinsightError: 当创建会话失败时
|
||||
"""
|
||||
try:
|
||||
# 生成唯一会话ID
|
||||
chat_id = uuid.uuid4().hex
|
||||
|
||||
# 创建消息会话
|
||||
message_session = MessageSession(
|
||||
chat_id=chat_id,
|
||||
flow_id='',
|
||||
flow_name='新对话',
|
||||
flow_type=FlowType.LINSIGHT.value,
|
||||
user_id=login_user.user_id
|
||||
)
|
||||
await MessageSessionDao.async_insert_one(message_session)
|
||||
|
||||
# 处理文件(如果存在)
|
||||
processed_files = await cls._process_submitted_files(submit_obj.files, chat_id)
|
||||
|
||||
# 创建灵思会话版本
|
||||
linsight_session_version = LinsightSessionVersion(
|
||||
session_id=chat_id,
|
||||
user_id=login_user.user_id,
|
||||
question=submit_obj.question,
|
||||
tools=submit_obj.tools,
|
||||
org_knowledge_enabled=submit_obj.org_knowledge_enabled,
|
||||
personal_knowledge_enabled=submit_obj.personal_knowledge_enabled,
|
||||
files=processed_files
|
||||
)
|
||||
linsight_session_version = await LinsightSessionVersionDao.insert_one(linsight_session_version)
|
||||
|
||||
return message_session, linsight_session_version
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"提交用户问题失败: {str(e)}")
|
||||
raise cls.LinsightError(f"提交用户问题失败: {str(e)}")
|
||||
|
||||
@classmethod
|
||||
async def _process_submitted_files(cls, files: Optional[List], chat_id: str) -> Optional[List]:
|
||||
"""
|
||||
处理提交的文件
|
||||
|
||||
Args:
|
||||
files: 文件列表
|
||||
chat_id: 会话ID
|
||||
|
||||
Returns:
|
||||
处理后的文件列表
|
||||
"""
|
||||
if not files:
|
||||
return None
|
||||
|
||||
file_ids = [file.file_id for file in files]
|
||||
redis_keys = [f"{cls.FILE_INFO_REDIS_KEY_PREFIX}{file_id}" for file_id in file_ids]
|
||||
|
||||
processed_files = await redis_client.amget(redis_keys)
|
||||
|
||||
for file_info in processed_files:
|
||||
if file_info:
|
||||
await cls._copy_file_to_session_storage(file_info, chat_id)
|
||||
|
||||
return processed_files
|
||||
|
||||
@classmethod
|
||||
async def _copy_file_to_session_storage(cls, file_info: Dict, chat_id: str) -> None:
|
||||
"""
|
||||
复制文件到会话存储
|
||||
|
||||
Args:
|
||||
file_info: 文件信息
|
||||
chat_id: 会话ID
|
||||
"""
|
||||
source_object_name = file_info.get("markdown_file_path")
|
||||
if source_object_name:
|
||||
original_filename = file_info.get("original_filename")
|
||||
markdown_filename = f"{original_filename.rsplit('.', 1)[0]}.md"
|
||||
new_object_name = f"linsight/{chat_id}/{source_object_name}"
|
||||
minio_client.copy_object(
|
||||
source_object_name=source_object_name,
|
||||
target_object_name=new_object_name,
|
||||
bucket_name=minio_client.tmp_bucket,
|
||||
target_bucket_name=minio_client.bucket
|
||||
)
|
||||
file_info["markdown_file_path"] = new_object_name
|
||||
file_info["markdown_filename"] = markdown_filename
|
||||
|
||||
@classmethod
|
||||
async def task_title_generate(cls, question: str, chat_id: str,
|
||||
login_user: UserPayload) -> Dict:
|
||||
"""
|
||||
生成任务标题
|
||||
|
||||
Args:
|
||||
question: 用户问题
|
||||
chat_id: 会话ID
|
||||
login_user: 登录用户信息
|
||||
|
||||
Returns:
|
||||
包含任务标题的字典
|
||||
"""
|
||||
try:
|
||||
# 获取并验证工作台配置
|
||||
workbench_conf = await cls._get_workbench_config()
|
||||
|
||||
# 创建LLM实例
|
||||
llm = BishengLLM(model_id=workbench_conf.task_model.id, temperature=0)
|
||||
|
||||
# 生成prompt
|
||||
prompt = await cls._generate_title_prompt(question)
|
||||
|
||||
# 生成任务标题
|
||||
task_title = await llm.ainvoke(prompt)
|
||||
|
||||
if not task_title.content:
|
||||
raise ValueError("生成任务标题失败,请检查模型配置或输入内容")
|
||||
|
||||
# 更新会话标题
|
||||
await cls._update_session_title(chat_id, task_title.content)
|
||||
|
||||
return {
|
||||
"task_title": task_title.content,
|
||||
"chat_id": chat_id,
|
||||
"error_message": None
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"生成任务标题失败: {str(e)}")
|
||||
return {
|
||||
"task_title": "新对话",
|
||||
"chat_id": chat_id,
|
||||
"error_message": str(e)
|
||||
}
|
||||
|
||||
@classmethod
|
||||
async def _get_workbench_config(cls):
|
||||
"""获取并验证工作台配置"""
|
||||
workbench_conf = await LLMService.get_workbench_llm()
|
||||
if not workbench_conf or not workbench_conf.task_model:
|
||||
raise cls.BishengLLMError("任务已终止,请联系管理员检查灵思任务执行模型状态")
|
||||
return workbench_conf
|
||||
|
||||
@classmethod
|
||||
async def _generate_title_prompt(cls, question: str) -> List[Tuple[str, str]]:
|
||||
"""生成标题生成的prompt"""
|
||||
prompt_service = app_ctx.get_prompt_loader()
|
||||
prompt_obj = prompt_service.render_prompt(
|
||||
namespace="gen_title",
|
||||
prompt_name="linsight",
|
||||
USER_GOAL=question
|
||||
)
|
||||
return [
|
||||
("system", prompt_obj.prompt.system),
|
||||
("user", prompt_obj.prompt.user)
|
||||
]
|
||||
|
||||
@classmethod
|
||||
async def _update_session_title(cls, chat_id: str, title: str) -> None:
|
||||
"""更新会话标题"""
|
||||
session = await MessageSessionDao.async_get_one(chat_id)
|
||||
if session:
|
||||
session.flow_name = title
|
||||
await MessageSessionDao.async_insert_one(session)
|
||||
|
||||
@classmethod
|
||||
async def get_linsight_session_version_list(cls, session_id: str) -> List[LinsightSessionVersion]:
|
||||
"""
|
||||
获取灵思会话版本列表
|
||||
|
||||
Args:
|
||||
session_id: 会话ID
|
||||
|
||||
Returns:
|
||||
灵思会话版本列表
|
||||
"""
|
||||
return await LinsightSessionVersionDao.get_session_versions_by_session_id(session_id)
|
||||
|
||||
@classmethod
|
||||
async def modify_sop(cls, linsight_session_version_id: str, sop_content: str) -> Dict:
|
||||
"""
|
||||
修改灵思会话版本的SOP内容
|
||||
|
||||
Args:
|
||||
linsight_session_version_id: 会话版本ID
|
||||
sop_content: SOP内容
|
||||
|
||||
Returns:
|
||||
操作结果
|
||||
"""
|
||||
try:
|
||||
await LinsightSessionVersionDao.modify_sop_content(
|
||||
linsight_session_version_id=linsight_session_version_id,
|
||||
sop_content=sop_content
|
||||
)
|
||||
return {"success": True, "message": "modify sop content successfully"}
|
||||
except Exception as e:
|
||||
logger.error(f"修改SOP内容失败: {str(e)}")
|
||||
return {"success": False, "message": str(e)}
|
||||
|
||||
@classmethod
|
||||
async def generate_sop(cls, linsight_session_version_id: str,
|
||||
previous_session_version_id: str,
|
||||
feedback_content: Optional[str] = None,
|
||||
reexecute: bool = False,
|
||||
login_user: Optional[UserPayload] = None,
|
||||
knowledge_list: List[KnowledgeRead] = None) -> AsyncGenerator[Dict, None]:
|
||||
"""
|
||||
生成SOP内容
|
||||
|
||||
Args:
|
||||
linsight_session_version_id: 当前会话版本ID
|
||||
previous_session_version_id: 上一个会话版本ID
|
||||
feedback_content: 反馈内容
|
||||
reexecute: 是否重新执行
|
||||
login_user: 登录用户信息
|
||||
knowledge_list: 知识库列表
|
||||
|
||||
Yields:
|
||||
生成的SOP内容事件
|
||||
"""
|
||||
error_message = None
|
||||
try:
|
||||
# 获取工作台配置和会话版本
|
||||
workbench_conf = await cls._get_workbench_config()
|
||||
session_version = await cls._get_session_version(linsight_session_version_id)
|
||||
|
||||
if login_user.user_id != session_version.user_id:
|
||||
yield {"event": "error", "data": "无权限操作该会话版本"}
|
||||
return
|
||||
try:
|
||||
# 创建LLM和工具
|
||||
llm = BishengLLM(model_id=workbench_conf.task_model.id, temperature=0)
|
||||
except Exception as e:
|
||||
logger.error(f"生成SOP内容失败: session_version_id={linsight_session_version_id}, error={str(e)}")
|
||||
raise cls.BishengLLMError(str(e))
|
||||
tools = await cls._prepare_tools(session_version, llm)
|
||||
|
||||
# 准备历史摘要
|
||||
history_summary = await cls._prepare_history_summary(
|
||||
reexecute, previous_session_version_id
|
||||
)
|
||||
|
||||
# 创建代理并生成SOP
|
||||
agent = await cls._create_linsight_agent(session_version, llm, tools, workbench_conf)
|
||||
|
||||
if previous_session_version_id:
|
||||
session_version = await LinsightSessionVersionDao.get_by_id(previous_session_version_id)
|
||||
|
||||
content = ""
|
||||
async for res in cls._generate_sop_content(
|
||||
agent, session_version, feedback_content, history_summary, knowledge_list
|
||||
):
|
||||
if isinstance(res, cls.SearchSOPError):
|
||||
yield {"event": "search_sop_error", "data": str(res.message)}
|
||||
continue
|
||||
|
||||
content += res.content
|
||||
yield {
|
||||
"event": "generate_sop_content",
|
||||
"data": res.model_dump_json()
|
||||
}
|
||||
|
||||
# 更新SOP内容
|
||||
await LinsightSessionVersionDao.modify_sop_content(
|
||||
linsight_session_version_id=linsight_session_version_id,
|
||||
sop_content=content
|
||||
)
|
||||
|
||||
logger.info(f"生成SOP内容成功: session_version_id={linsight_session_version_id}")
|
||||
|
||||
|
||||
except cls.ToolsInitializationError as e:
|
||||
logger.exception(
|
||||
f"初始化灵思工作台工具失败: session_version_id={linsight_session_version_id}, error={str(e)}")
|
||||
error_message = f"初始化灵思工作台工具失败: {str(e)}"
|
||||
except cls.BishengLLMError as e:
|
||||
logger.exception(f"Bisheng LLM错误: session_version_id={linsight_session_version_id}, error={str(e)}")
|
||||
error_message = str(e)
|
||||
except Exception as e:
|
||||
logger.exception(f"生成SOP内容失败: session_version_id={linsight_session_version_id}, error={str(e)}")
|
||||
error_message = f"生成SOP内容失败: {str(e)}"
|
||||
|
||||
finally:
|
||||
if error_message:
|
||||
session_version = await LinsightSessionVersionDao.get_by_id(linsight_session_version_id)
|
||||
if session_version:
|
||||
session_version.sop = error_message
|
||||
session_version.status = SessionVersionStatusEnum.SOP_GENERATION_FAILED
|
||||
await LinsightSessionVersionDao.insert_one(session_version)
|
||||
yield {"event": "error", "data": error_message}
|
||||
|
||||
@classmethod
|
||||
async def _get_session_version(cls, session_version_id: str) -> LinsightSessionVersion:
|
||||
"""获取会话版本"""
|
||||
session_version = await LinsightSessionVersionDao.get_by_id(session_version_id)
|
||||
if not session_version:
|
||||
raise cls.LinsightError("灵思会话版本不存在")
|
||||
return session_version
|
||||
|
||||
@classmethod
|
||||
async def _prepare_tools(cls, session_version: LinsightSessionVersion,
|
||||
llm: BishengLLM) -> List[BaseTool]:
|
||||
"""准备工具列表"""
|
||||
try:
|
||||
tools = await cls.init_linsight_config_tools(session_version, llm)
|
||||
|
||||
root_path = os.path.join(CACHE_DIR, "linsight", session_version.id)
|
||||
os.makedirs(root_path, exist_ok=True)
|
||||
|
||||
linsight_tools = await ToolServices.init_linsight_tools(root_path=root_path)
|
||||
tools.extend(linsight_tools)
|
||||
|
||||
return tools
|
||||
except Exception as e:
|
||||
raise cls.ToolsInitializationError(f"初始化灵思工作台工具失败: {str(e)}")
|
||||
|
||||
@classmethod
|
||||
async def _prepare_file_list(cls, session_version: LinsightSessionVersion) -> List[str]:
|
||||
"""准备文件列表"""
|
||||
file_list = []
|
||||
template_str = """@{filename}的文件储存信息:{{"文件储存在语义检索库中的id":"{file_id}","文件储存地址":"{markdown}"}}@"""
|
||||
if not session_version.files:
|
||||
return file_list
|
||||
for file in session_version.files:
|
||||
file_list.append(template_str.format(filename=file['original_filename'],
|
||||
file_id=file['file_id'],
|
||||
markdown=f"./{file['markdown_filename']}"))
|
||||
return file_list
|
||||
|
||||
@classmethod
|
||||
async def _prepare_knowledge_list(cls, knowledge_list: list[KnowledgeRead]) -> List[str]:
|
||||
res = []
|
||||
if not knowledge_list:
|
||||
return res
|
||||
# 查询是否有个人知识库
|
||||
template_str = """@{name}的储存信息:{{"知识库储存在语义检索库中的id":"{id}"}}@"""
|
||||
for one in knowledge_list:
|
||||
if one.type == KnowledgeTypeEnum.PRIVATE.value:
|
||||
res.append(template_str.format(name="个人知识库", id=one.id))
|
||||
else:
|
||||
knowledge_str = template_str.format(name=one.name, id=one.id)
|
||||
if one.description:
|
||||
knowledge_str += f",{one.name}的描述是{one.description}"
|
||||
res.append(knowledge_str)
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
async def _prepare_history_summary(cls, reexecute: bool,
|
||||
previous_session_version_id: str) -> List[str]:
|
||||
"""准备历史摘要"""
|
||||
history_summary = []
|
||||
|
||||
if reexecute and previous_session_version_id:
|
||||
execute_tasks = await LinsightExecuteTaskDao.get_by_session_version_id(previous_session_version_id)
|
||||
|
||||
for task in execute_tasks:
|
||||
if task.result:
|
||||
answer = task.result.get("answer", "")
|
||||
if answer:
|
||||
history_summary.append(answer)
|
||||
|
||||
return history_summary
|
||||
|
||||
@classmethod
|
||||
async def _create_linsight_agent(cls, session_version: LinsightSessionVersion,
|
||||
llm: BishengLLM, tools: List[BaseTool],
|
||||
workbench_conf):
|
||||
"""创建Linsight代理"""
|
||||
from bisheng_langchain.linsight.agent import LinsightAgent
|
||||
|
||||
root_path = os.path.join(CACHE_DIR, "linsight", session_version.id[:8])
|
||||
linsight_conf = settings.get_linsight_conf()
|
||||
exec_config = ExecConfig(**linsight_conf.model_dump(), debug_id=session_version.id)
|
||||
return LinsightAgent(
|
||||
file_dir=root_path,
|
||||
query=session_version.question,
|
||||
llm=llm,
|
||||
tools=tools,
|
||||
task_mode=workbench_conf.linsight_executor_mode,
|
||||
exec_config=exec_config,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
async def _generate_sop_content(cls, agent, session_version: LinsightSessionVersion,
|
||||
feedback_content: Optional[str],
|
||||
history_summary: List[str],
|
||||
knowledge_list: List[KnowledgeRead] = None) -> AsyncGenerator:
|
||||
"""生成SOP内容"""
|
||||
file_list = await cls._prepare_file_list(session_version)
|
||||
knowledge_list = await cls._prepare_knowledge_list(knowledge_list)
|
||||
|
||||
if feedback_content is None:
|
||||
# 检索SOP模板
|
||||
sop_template, search_sop_error_msg = await SOPManageService.search_sop(
|
||||
query=session_version.question, k=3
|
||||
)
|
||||
|
||||
if search_sop_error_msg:
|
||||
logger.error(f"检索SOP模板失败: {search_sop_error_msg}")
|
||||
yield cls.SearchSOPError(message=search_sop_error_msg)
|
||||
|
||||
sop_template = "\n\n".join([
|
||||
f"例子:\n\n{sop.page_content}"
|
||||
for sop in sop_template if sop.page_content
|
||||
])
|
||||
|
||||
async for res in agent.generate_sop(sop=sop_template, file_list=file_list, knowledge_list=knowledge_list):
|
||||
yield res
|
||||
else:
|
||||
|
||||
sop_template = session_version.sop if session_version.sop else ""
|
||||
if sop_template:
|
||||
sop_template = f"例子:\n\n{sop_template}"
|
||||
|
||||
async for res in agent.feedback_sop(
|
||||
sop=sop_template,
|
||||
feedback=feedback_content,
|
||||
history_summary=history_summary if history_summary else None,
|
||||
file_list=file_list,
|
||||
knowledge_list=knowledge_list
|
||||
):
|
||||
yield res
|
||||
|
||||
@classmethod
|
||||
async def get_execute_task_detail(cls, session_version_id: str,
|
||||
login_user: Optional[UserPayload] = None):
|
||||
"""
|
||||
获取执行任务详情
|
||||
|
||||
Args:
|
||||
session_version_id: 灵思会话版本ID
|
||||
login_user: 登录用户信息
|
||||
|
||||
Returns:
|
||||
执行任务详情列表
|
||||
"""
|
||||
execute_tasks = await LinsightExecuteTaskDao.get_by_session_version_id(session_version_id)
|
||||
|
||||
if not execute_tasks:
|
||||
return []
|
||||
|
||||
# 1. 获取一级任务 parent_task_id 是 None 的任务
|
||||
root_tasks = [task for task in execute_tasks if task.parent_task_id is None]
|
||||
|
||||
# 2. 根据previous_task_id与next_task_id排序一级任务
|
||||
def sort_tasks_by_chain(tasks: List[Any]) -> List[Any]:
|
||||
"""
|
||||
根据任务链排序任务列表
|
||||
previous_task_id是None则是第一个任务,next_task_id是None则是最后一个任务
|
||||
"""
|
||||
if not tasks:
|
||||
return []
|
||||
|
||||
# 创建任务字典以便快速查找
|
||||
task_dict = {task.id: task for task in tasks}
|
||||
|
||||
# 找到链的开始节点(previous_task_id 为 None)
|
||||
start_tasks = [task for task in tasks if task.previous_task_id is None]
|
||||
|
||||
sorted_tasks = []
|
||||
|
||||
for start_task in start_tasks:
|
||||
# 从每个开始节点构建任务链
|
||||
current_task = start_task
|
||||
chain = []
|
||||
|
||||
while current_task is not None:
|
||||
chain.append(current_task)
|
||||
# 通过next_task_id找到下一个任务
|
||||
next_task_id = current_task.next_task_id
|
||||
current_task = task_dict.get(next_task_id) if next_task_id else None
|
||||
|
||||
sorted_tasks.extend(chain)
|
||||
|
||||
# 处理可能存在的孤立任务(既没有previous也没有next指向它们)
|
||||
processed_ids = {task.id for task in sorted_tasks}
|
||||
orphan_tasks = [task for task in tasks if task.id not in processed_ids]
|
||||
sorted_tasks.extend(orphan_tasks)
|
||||
|
||||
return sorted_tasks
|
||||
|
||||
# 排序一级任务
|
||||
sorted_root_tasks = sort_tasks_by_chain(root_tasks)
|
||||
|
||||
# 3. 构建任务树 使用 parent_task_id 将子任务与父任务关联起来
|
||||
def build_task_tree(parent_tasks: List[Any], all_tasks: List[Any]) -> List[TaskNode]:
|
||||
"""
|
||||
构建任务树
|
||||
"""
|
||||
# 创建任务映射
|
||||
task_map = {task.id: task for task in all_tasks}
|
||||
|
||||
# 按父任务ID分组子任务
|
||||
children_map = {}
|
||||
for task in all_tasks:
|
||||
if task.parent_task_id:
|
||||
if task.parent_task_id not in children_map:
|
||||
children_map[task.parent_task_id] = []
|
||||
children_map[task.parent_task_id].append(task)
|
||||
|
||||
def build_node(task: Any) -> TaskNode:
|
||||
"""递归构建任务节点"""
|
||||
node = TaskNode(task=task)
|
||||
|
||||
# 获取子任务
|
||||
child_tasks = children_map.get(task.id, [])
|
||||
|
||||
# 对子任务进行排序
|
||||
sorted_child_tasks = sort_tasks_by_chain(child_tasks)
|
||||
|
||||
# 递归构建子节点
|
||||
for child_task in sorted_child_tasks:
|
||||
child_node = build_node(child_task)
|
||||
node.children.append(child_node)
|
||||
|
||||
return node
|
||||
|
||||
# 构建根节点列表
|
||||
root_nodes = []
|
||||
for parent_task in parent_tasks:
|
||||
root_node = build_node(parent_task)
|
||||
root_nodes.append(root_node)
|
||||
|
||||
return root_nodes
|
||||
|
||||
# 构建任务树
|
||||
task_tree = build_task_tree(sorted_root_tasks, execute_tasks)
|
||||
|
||||
# 4. 返回任务树的根节点列表
|
||||
result = [node.to_dict() for node in task_tree]
|
||||
|
||||
return result
|
||||
|
||||
@classmethod
|
||||
async def upload_file(cls, file: UploadFile) -> Dict:
|
||||
"""
|
||||
上传文件到灵思工作台
|
||||
|
||||
Args:
|
||||
file: 上传的文件
|
||||
|
||||
Returns:
|
||||
文件信息字典
|
||||
"""
|
||||
# 生成文件信息
|
||||
file_id = uuid.uuid4().hex[:8] # 生成8位唯一文件ID
|
||||
# url 编码 decode 文件名
|
||||
original_filename = unquote(file.filename)
|
||||
file_extension = original_filename.split('.')[-1] if '.' in original_filename else ''
|
||||
unique_filename = f"{file_id}.{file_extension}"
|
||||
|
||||
# 保存文件
|
||||
file_path = await save_file_to_folder(file, 'linsight', unique_filename)
|
||||
|
||||
upload_result = {
|
||||
"file_id": file_id,
|
||||
"filename": unique_filename,
|
||||
"original_filename": original_filename,
|
||||
"file_path": file_path,
|
||||
"parsing_status": "running",
|
||||
}
|
||||
|
||||
# 缓存解析结果
|
||||
await cls._cache_parse_result(file_id, upload_result)
|
||||
|
||||
return upload_result
|
||||
|
||||
@classmethod
|
||||
async def parse_file(cls, upload_result: Dict) -> Dict:
|
||||
"""
|
||||
解析上传的文件
|
||||
|
||||
Args:
|
||||
upload_result: 上传结果
|
||||
|
||||
Returns:
|
||||
解析结果
|
||||
"""
|
||||
logger.info(f"开始解析文件: {upload_result}")
|
||||
|
||||
file_id = upload_result["file_id"]
|
||||
original_filename = upload_result["original_filename"]
|
||||
file_path = upload_result["file_path"]
|
||||
try:
|
||||
# 获取工作台配置
|
||||
workbench_conf = await cls._get_workbench_config()
|
||||
collection_name = f"{cls.COLLECTION_NAME_PREFIX}{workbench_conf.embedding_model.id}"
|
||||
|
||||
# 异步执行文件解析
|
||||
parse_result = await util.sync_func_to_async(cls._parse_file_sync)(file_id, file_path, original_filename,
|
||||
collection_name, workbench_conf)
|
||||
|
||||
# 缓存解析结果
|
||||
await cls._cache_parse_result(file_id, parse_result)
|
||||
|
||||
logger.info(f"文件解析完成: {parse_result}")
|
||||
except Exception as e:
|
||||
logger.error(f"文件解析失败: file_id={file_id}, error={str(e)}")
|
||||
parse_result = {
|
||||
"file_id": file_id,
|
||||
"original_filename": original_filename,
|
||||
"parsing_status": "failed",
|
||||
"error_message": str(e)
|
||||
}
|
||||
await cls._cache_parse_result(file_id, parse_result)
|
||||
|
||||
return parse_result
|
||||
|
||||
@classmethod
|
||||
def _parse_file_sync(cls, file_id: str, file_path: str, original_filename: str,
|
||||
collection_name: str, workbench_conf) -> Dict:
|
||||
"""
|
||||
同步解析文件
|
||||
|
||||
Args:
|
||||
file_id: 文件ID
|
||||
file_path: 文件路径
|
||||
original_filename: 原始文件名
|
||||
collection_name: 集合名称
|
||||
workbench_conf: 工作台配置
|
||||
|
||||
Returns:
|
||||
解析结果
|
||||
"""
|
||||
# 读取文件内容
|
||||
try:
|
||||
texts, _, parse_type, _ = read_chunk_text(
|
||||
input_file=file_path,
|
||||
file_name=original_filename,
|
||||
separator=['\n\n', '\n'],
|
||||
separator_rule=['after', 'after'],
|
||||
chunk_size=1000,
|
||||
chunk_overlap=100,
|
||||
no_summary=True
|
||||
)
|
||||
|
||||
# 生成markdown内容
|
||||
markdown_content = "\n".join(texts)
|
||||
markdown_bytes = markdown_content.encode('utf-8')
|
||||
|
||||
# 保存markdown文件
|
||||
markdown_filename = f"{file_id}.md"
|
||||
minio_client.upload_tmp(markdown_filename, markdown_bytes)
|
||||
markdown_md5 = calculate_md5(markdown_bytes)
|
||||
|
||||
# 处理向量存储
|
||||
cls._process_vector_storage(texts, file_id, collection_name, workbench_conf)
|
||||
|
||||
return {
|
||||
"file_id": file_id,
|
||||
"original_filename": original_filename,
|
||||
"parsing_status": "completed",
|
||||
"parse_type": parse_type,
|
||||
"markdown_filename": markdown_filename,
|
||||
"markdown_file_path": markdown_filename,
|
||||
"markdown_file_md5": markdown_md5,
|
||||
"embedding_model_id": workbench_conf.embedding_model.id,
|
||||
"collection_name": collection_name
|
||||
}
|
||||
except Exception as e:
|
||||
logger.error(f"文件解析失败: file_id={file_id}, error={str(e)}")
|
||||
return {
|
||||
"file_id": file_id,
|
||||
"original_filename": original_filename,
|
||||
"parsing_status": "failed",
|
||||
"error_message": str(e)
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def _process_vector_storage(cls, texts: List[str], file_id: str,
|
||||
collection_name: str, workbench_conf) -> None:
|
||||
"""处理向量存储"""
|
||||
# 创建embeddings
|
||||
embeddings = decide_embeddings(workbench_conf.embedding_model.id)
|
||||
|
||||
# 创建向量存储
|
||||
vector_client = decide_vectorstores(collection_name, "Milvus", embeddings)
|
||||
es_client = decide_vectorstores(collection_name, "ElasticKeywordsSearch", FakeEmbedding())
|
||||
|
||||
# 添加文本到向量存储
|
||||
metadatas = [{"file_id": file_id} for _ in texts]
|
||||
vector_client.add_texts(texts, metadatas=metadatas)
|
||||
es_client.add_texts(texts, metadatas=metadatas)
|
||||
|
||||
@classmethod
|
||||
async def _cache_parse_result(cls, file_id: str, parse_result: Dict) -> None:
|
||||
"""缓存解析结果"""
|
||||
key = f"{cls.FILE_INFO_REDIS_KEY_PREFIX}{file_id}"
|
||||
await redis_client.aset(
|
||||
key=key,
|
||||
value=parse_result,
|
||||
expiration=60 * 60 * cls.CACHE_EXPIRATION_HOURS
|
||||
)
|
||||
|
||||
@classmethod
|
||||
async def init_linsight_config_tools(cls, session_version: LinsightSessionVersion,
|
||||
llm: BishengLLM) -> List[BaseTool]:
|
||||
"""
|
||||
初始化灵思配置的工具
|
||||
|
||||
Args:
|
||||
session_version: 会话版本模型
|
||||
llm: LLM实例
|
||||
|
||||
Returns:
|
||||
工具列表
|
||||
"""
|
||||
tools = []
|
||||
|
||||
if not session_version.tools:
|
||||
return tools
|
||||
|
||||
# 提取工具ID
|
||||
tool_ids = cls._extract_tool_ids(session_version.tools)
|
||||
|
||||
# 获取工作台配置的工具ID
|
||||
ws_config = await WorkStationService.aget_config()
|
||||
config_tool_ids = cls._extract_tool_ids(ws_config.linsightConfig.tools or [])
|
||||
|
||||
# 过滤有效的工具ID
|
||||
valid_tool_ids = [tid for tid in tool_ids if tid in config_tool_ids]
|
||||
|
||||
# 初始化工具
|
||||
if valid_tool_ids:
|
||||
tools.extend(await AssistantAgent.init_tools_by_tool_ids(valid_tool_ids, llm=llm))
|
||||
|
||||
return tools
|
||||
|
||||
@classmethod
|
||||
def _extract_tool_ids(cls, tools: List[Dict]) -> List[int]:
|
||||
"""
|
||||
从工具配置中提取工具ID
|
||||
|
||||
Args:
|
||||
tools: 工具配置列表
|
||||
|
||||
Returns:
|
||||
工具ID列表
|
||||
"""
|
||||
tool_ids = []
|
||||
for tool in tools:
|
||||
if tool.get("children"):
|
||||
tool_ids.extend(int(child.get("id")) for child in tool["children"] if child.get("id"))
|
||||
return tool_ids
|
||||
|
||||
@classmethod
|
||||
async def feedback_regenerate_sop_task(cls, session_version_model: LinsightSessionVersion,
|
||||
feedback: str) -> None:
|
||||
"""
|
||||
根据反馈重新生成SOP任务
|
||||
|
||||
Args:
|
||||
session_version_model: 灵思会话版本模型
|
||||
feedback: 反馈内容
|
||||
"""
|
||||
try:
|
||||
file_list = await cls._prepare_file_list(session_version_model)
|
||||
# 获取工作台配置
|
||||
workbench_conf = await cls._get_workbench_config()
|
||||
|
||||
# 创建LLM和工具
|
||||
llm = BishengLLM(model_id=workbench_conf.task_model.id, temperature=0)
|
||||
tools = await cls._prepare_tools(session_version_model, llm)
|
||||
|
||||
# 获取历史摘要
|
||||
history_summary = await cls._get_history_summary(session_version_model.id)
|
||||
|
||||
# 创建代理并生成SOP
|
||||
agent = await cls._create_linsight_agent(session_version_model, llm, tools, workbench_conf)
|
||||
|
||||
sop_content = ""
|
||||
sop_template = f"例子:\n\n{session_version_model.sop or ''}"
|
||||
|
||||
async for res in agent.feedback_sop(
|
||||
sop=sop_template,
|
||||
feedback=feedback,
|
||||
history_summary=history_summary if history_summary else None,
|
||||
file_list=file_list
|
||||
):
|
||||
sop_content += res.content
|
||||
|
||||
# sop写到记录表里,这个sop不需要关联会话,因为不需要更新分数
|
||||
await SOPManageService.add_sop_record(LinsightSOPRecord(
|
||||
name=session_version_model.title,
|
||||
description=None,
|
||||
user_id=session_version_model.user_id,
|
||||
content=sop_content,
|
||||
))
|
||||
except cls.ToolsInitializationError as e:
|
||||
logger.exception(f"初始化灵思工作台工具失败: session_version_id={session_version_model.id}, error={str(e)}")
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"反馈重新生成SOP任务失败: session_version_id={session_version_model.id}, error={str(e)}")
|
||||
|
||||
@classmethod
|
||||
async def _get_history_summary(cls, session_version_id: str) -> List[str]:
|
||||
"""获取历史摘要"""
|
||||
history_summary = []
|
||||
execute_tasks = await LinsightExecuteTaskDao.get_by_session_version_id(session_version_id)
|
||||
|
||||
for task in execute_tasks:
|
||||
if task.result:
|
||||
answer = task.result.get("answer", "")
|
||||
if answer:
|
||||
history_summary.append(answer)
|
||||
|
||||
return history_summary
|
||||
|
||||
@classmethod
|
||||
async def batch_download_files(cls, file_info_list: List[BatchDownloadFilesSchema]) -> bytes:
|
||||
"""
|
||||
批量下载文件
|
||||
|
||||
Args:
|
||||
file_info_list: 文件信息列表
|
||||
|
||||
Returns:
|
||||
包含文件下载信息的列表
|
||||
"""
|
||||
|
||||
async def download_file(file_info: BatchDownloadFilesSchema) -> Tuple[str, bytes]:
|
||||
"""下载单个文件"""
|
||||
object_name = file_info.file_url
|
||||
try:
|
||||
|
||||
bytes_io = BytesIO()
|
||||
|
||||
file_byte = await util.sync_func_to_async(minio_client.get_object)(bucket_name=minio_client.bucket,
|
||||
object_name=object_name)
|
||||
bytes_io.write(file_byte)
|
||||
|
||||
bytes_io.seek(0)
|
||||
|
||||
return file_info.file_name, bytes_io.getvalue()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"下载文件失败 {object_name}: {e}")
|
||||
return object_name, b''
|
||||
|
||||
# 批量下载文件
|
||||
download_tasks = [download_file(file_info) for file_info in file_info_list]
|
||||
|
||||
results = await asyncio.gather(*download_tasks)
|
||||
|
||||
# 过滤掉下载失败的文件
|
||||
successful_files = [res for res in results if res[1]]
|
||||
|
||||
if not successful_files:
|
||||
raise ValueError("没有成功下载的文件,无法生成ZIP")
|
||||
|
||||
zip_bytes = util.bytes_to_zip(successful_files)
|
||||
return zip_bytes
|
||||
@@ -0,0 +1,453 @@
|
||||
import json
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import Request, BackgroundTasks
|
||||
from langchain_core.embeddings import Embeddings
|
||||
from langchain_core.language_models import BaseChatModel
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.errcode.base import NotFoundError
|
||||
from bisheng.api.errcode.llm import ServerExistError, ModelNameRepeatError, ServerAddError, ServerAddAllError
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.v1.schemas import LLMServerInfo, LLMModelInfo, KnowledgeLLMConfig, AssistantLLMConfig, \
|
||||
EvaluationLLMConfig, AssistantLLMItem, LLMServerCreateReq, WorkbenchModelConfig
|
||||
from bisheng.database.models.config import ConfigDao, ConfigKeyEnum, Config
|
||||
from bisheng.database.models.llm_server import LLMDao, LLMServer, LLMModel, LLMModelType
|
||||
from bisheng.interface.importing import import_by_type
|
||||
from bisheng.interface.initialize.loading import instantiate_llm, instantiate_embedding
|
||||
from bisheng.utils.embedding import decide_embeddings
|
||||
|
||||
|
||||
class LLMService:
|
||||
|
||||
@classmethod
|
||||
def get_all_llm(cls, request: Request, login_user: UserPayload) -> List[LLMServerInfo]:
|
||||
""" 获取所有的模型数据, 不包含key等敏感信息 """
|
||||
llm_servers = LLMDao.get_all_server()
|
||||
ret = []
|
||||
server_ids = []
|
||||
for one in llm_servers:
|
||||
server_ids.append(one.id)
|
||||
ret.append(LLMServerInfo(**one.model_dump(exclude={'config'})))
|
||||
|
||||
llm_models = LLMDao.get_model_by_server_ids(server_ids)
|
||||
server_dicts = {}
|
||||
for one in llm_models:
|
||||
if one.server_id not in server_dicts:
|
||||
server_dicts[one.server_id] = []
|
||||
server_dicts[one.server_id].append(LLMModelInfo(**one.model_dump(exclude={'config'})))
|
||||
|
||||
for one in ret:
|
||||
one.models = server_dicts.get(one.id, [])
|
||||
return ret
|
||||
|
||||
@classmethod
|
||||
def get_one_llm(cls, request: Request, login_user: UserPayload, server_id: int) -> LLMServerInfo:
|
||||
""" 获取一个服务提供方的详细信息 包含了key等敏感的配置信息 """
|
||||
llm = LLMDao.get_server_by_id(server_id)
|
||||
if not llm:
|
||||
raise NotFoundError.http_exception()
|
||||
|
||||
models = LLMDao.get_model_by_server_ids([server_id])
|
||||
models = [LLMModelInfo(**one.model_dump()) for one in models]
|
||||
return LLMServerInfo(**llm.model_dump(), models=models)
|
||||
|
||||
@classmethod
|
||||
def add_llm_server(cls, request: Request, login_user: UserPayload, server: LLMServerCreateReq) -> LLMServerInfo:
|
||||
""" 添加一个服务提供方 """
|
||||
exist_server = LLMDao.get_server_by_name(server.name)
|
||||
if exist_server:
|
||||
raise ServerExistError.http_exception()
|
||||
|
||||
model_dict = {}
|
||||
for one in server.models:
|
||||
if one.model_name not in model_dict:
|
||||
model_dict[one.model_name] = LLMModel(**one.dict(), user_id=login_user.user_id)
|
||||
else:
|
||||
raise ModelNameRepeatError.http_exception()
|
||||
|
||||
db_server = LLMServer(**server.dict(exclude={'models'}))
|
||||
db_server.user_id = login_user.user_id
|
||||
|
||||
db_server = LLMDao.insert_server_with_models(db_server, list(model_dict.values()))
|
||||
|
||||
ret = cls.get_one_llm(request, login_user, db_server.id)
|
||||
success_models = []
|
||||
success_msg = ''
|
||||
failed_models = []
|
||||
failed_msg = ''
|
||||
# 尝试实例化对应的模型,有报错的话删除
|
||||
for one in ret.models:
|
||||
try:
|
||||
if one.model_type == LLMModelType.LLM.value:
|
||||
cls.get_bisheng_llm(model_id=one.id, ignore_online=True)
|
||||
elif one.model_type == LLMModelType.EMBEDDING.value:
|
||||
cls.get_bisheng_embedding(model_id=one.id, ignore_online=True)
|
||||
success_msg += f'{one.model_name},'
|
||||
success_models.append(one)
|
||||
except Exception as e:
|
||||
logger.exception("init_model_error")
|
||||
# 模型初始化失败的话,不添加到模型列表里
|
||||
failed_msg += f'<{one.model_name}>添加失败,失败原因:{str(e)}\n'
|
||||
failed_models.append(one)
|
||||
|
||||
# 说明模型全部添加失败了
|
||||
if len(success_models) == 0 and failed_msg:
|
||||
LLMDao.delete_server_by_id(ret.id)
|
||||
raise ServerAddAllError.http_exception(failed_msg)
|
||||
elif len(success_models) > 0 and failed_msg:
|
||||
# 部分模型添加成功了, 删除失败的模型信息
|
||||
ret.models = success_models
|
||||
LLMDao.delete_model_by_ids(model_ids=[one.id for one in failed_models])
|
||||
cls.add_llm_server_hook(request, login_user, ret)
|
||||
raise ServerAddError.http_exception(f"<{success_msg.rstrip(',')}>添加成功,{failed_msg}")
|
||||
|
||||
cls.add_llm_server_hook(request, login_user, ret)
|
||||
return ret
|
||||
|
||||
@classmethod
|
||||
def delete_llm_server(cls, request: Request, login_user: UserPayload, server_id: int) -> bool:
|
||||
""" 删除一个服务提供方 """
|
||||
LLMDao.delete_server_by_id(server_id)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def add_llm_server_hook(cls, request: Request, login_user: UserPayload, server: LLMServerInfo) -> bool:
|
||||
""" 添加一个服务提供方 后续动作 """
|
||||
|
||||
handle_types = []
|
||||
for one in server.models:
|
||||
# test model status
|
||||
cls.test_model_status(one)
|
||||
if one.model_type in handle_types:
|
||||
continue
|
||||
handle_types.append(one.model_type)
|
||||
model_info = LLMDao.get_model_by_type(LLMModelType(one.model_type))
|
||||
# 判断是否是首个llm或者embedding模型
|
||||
if model_info.id == one.id:
|
||||
cls.set_default_model(request, login_user, model_info)
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def test_model_status(cls, model: LLMModel | LLMModelInfo):
|
||||
try:
|
||||
if model.model_type == LLMModelType.LLM.value:
|
||||
bisheng_model = cls.get_bisheng_llm(model_id=model.id, ignore_online=True, cache=False)
|
||||
bisheng_model.invoke('hello')
|
||||
elif model.model_type == LLMModelType.EMBEDDING.value:
|
||||
bisheng_embed = cls.get_bisheng_embedding(model_id=model.id, ignore_online=True, cache=False)
|
||||
bisheng_embed.embed_query('hello')
|
||||
except Exception as e:
|
||||
LLMDao.update_model_status(model.id, 1, str(e))
|
||||
logger.exception(f'test model status: {model.id} {model.model_name}')
|
||||
|
||||
@classmethod
|
||||
def set_default_model(cls, request: Request, login_user: UserPayload, model: LLMModel):
|
||||
""" 设置默认的模型配置 """
|
||||
# 设置默认的llm模型配置
|
||||
if model.model_type == LLMModelType.LLM.value:
|
||||
# 设置知识库的默认模型配置
|
||||
knowledge_llm = cls.get_knowledge_llm()
|
||||
knowledge_change = False
|
||||
if not knowledge_llm.extract_title_model_id:
|
||||
knowledge_llm.extract_title_model_id = model.id
|
||||
knowledge_change = True
|
||||
if not knowledge_llm.source_model_id:
|
||||
knowledge_llm.source_model_id = model.id
|
||||
knowledge_change = True
|
||||
if not knowledge_llm.qa_similar_model_id:
|
||||
knowledge_llm.qa_similar_model_id = model.id
|
||||
knowledge_change = True
|
||||
if knowledge_change:
|
||||
cls.update_knowledge_llm(request, login_user, knowledge_llm)
|
||||
|
||||
# 设置评测的默认模型配置
|
||||
evaluation_llm = cls.get_evaluation_llm()
|
||||
if not evaluation_llm.model_id:
|
||||
evaluation_llm.model_id = model.id
|
||||
cls.update_evaluation_llm(request, login_user, evaluation_llm)
|
||||
|
||||
# 设置助手的默认模型配置
|
||||
assistant_llm = cls.get_assistant_llm()
|
||||
assistant_change = False
|
||||
if not assistant_llm.auto_llm:
|
||||
assistant_llm.auto_llm = AssistantLLMItem(model_id=model.id)
|
||||
assistant_change = True
|
||||
if not assistant_llm.llm_list:
|
||||
assistant_change = True
|
||||
assistant_llm.llm_list = [
|
||||
AssistantLLMItem(model_id=model.id, default=True)
|
||||
]
|
||||
if assistant_change:
|
||||
cls.update_assistant_llm(request, login_user, assistant_llm)
|
||||
|
||||
elif model.model_type == LLMModelType.EMBEDDING.value:
|
||||
knowledge_llm = cls.get_knowledge_llm()
|
||||
if not knowledge_llm.embedding_model_id:
|
||||
knowledge_llm.embedding_model_id = model.id
|
||||
cls.update_knowledge_llm(request, login_user, knowledge_llm)
|
||||
|
||||
@classmethod
|
||||
def update_llm_server(cls, request: Request, login_user: UserPayload, server: LLMServerCreateReq) -> LLMServerInfo:
|
||||
""" 更新服务提供方信息 """
|
||||
exist_server = LLMDao.get_server_by_id(server.id)
|
||||
if not exist_server:
|
||||
raise NotFoundError.http_exception()
|
||||
|
||||
old_models = LLMDao.get_model_by_server_ids([exist_server.id])
|
||||
old_model_dict = {
|
||||
one.id: one for one in old_models
|
||||
}
|
||||
if exist_server.name != server.name:
|
||||
# 改名的话判断下是否已经存在
|
||||
name_server = LLMDao.get_server_by_name(server.name)
|
||||
if name_server and name_server.id != server.id:
|
||||
raise ServerExistError.http_exception(f'<{server.name}>已存在')
|
||||
|
||||
model_dict = {}
|
||||
for one in server.models:
|
||||
if one.model_name not in model_dict:
|
||||
model_dict[one.model_name] = LLMModel(**one.model_dump())
|
||||
# 说明是新增模型
|
||||
if not one.id:
|
||||
model_dict[one.model_name].user_id = login_user.user_id
|
||||
model_dict[one.model_name].server_id = exist_server.id
|
||||
else:
|
||||
raise ModelNameRepeatError.http_exception()
|
||||
|
||||
exist_server.name = server.name
|
||||
exist_server.description = server.description
|
||||
exist_server.type = server.type
|
||||
exist_server.limit_flag = server.limit_flag
|
||||
exist_server.limit = server.limit
|
||||
exist_server.config = server.config
|
||||
|
||||
db_server = LLMDao.update_server_with_models(exist_server, list(model_dict.values()))
|
||||
new_server_info = cls.get_one_llm(request, login_user, db_server.id)
|
||||
|
||||
# 判断是否需要重新判断模型状态
|
||||
for one in new_server_info.models:
|
||||
# 新增的模型,或者模型名字或者类型发生了变化
|
||||
if (one.id not in old_model_dict or old_model_dict[one.id].model_name != one.model_name
|
||||
or old_model_dict[one.id].model_type != one.model_type):
|
||||
cls.test_model_status(one)
|
||||
return new_server_info
|
||||
|
||||
@classmethod
|
||||
def update_model_online(cls, request: Request, login_user: UserPayload, model_id: int,
|
||||
online: bool) -> LLMModelInfo:
|
||||
""" 更新模型是否上线 """
|
||||
exist_model = LLMDao.get_model_by_id(model_id)
|
||||
if not exist_model:
|
||||
raise NotFoundError.http_exception()
|
||||
exist_model.online = online
|
||||
LLMDao.update_model_online(exist_model.id, online)
|
||||
return LLMModelInfo(**exist_model.dict())
|
||||
|
||||
@classmethod
|
||||
def get_knowledge_llm(cls) -> KnowledgeLLMConfig:
|
||||
""" 获取知识库相关的默认模型配置 """
|
||||
ret = {}
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.KNOWLEDGE_LLM)
|
||||
if config:
|
||||
ret = json.loads(config.value)
|
||||
return KnowledgeLLMConfig(**ret)
|
||||
|
||||
@classmethod
|
||||
def get_knowledge_source_llm(cls) -> Optional[BaseChatModel]:
|
||||
""" 获取知识库溯源的默认模型配置 """
|
||||
knowledge_llm = cls.get_knowledge_llm()
|
||||
# 没有配置模型,则用jieba
|
||||
if not knowledge_llm.source_model_id:
|
||||
return None
|
||||
return cls.get_bisheng_llm(model_id=knowledge_llm.source_model_id)
|
||||
|
||||
@classmethod
|
||||
def get_knowledge_similar_llm(cls) -> Optional[BaseChatModel]:
|
||||
""" 获取知识库相似问的默认模型配置 """
|
||||
knowledge_llm = cls.get_knowledge_llm()
|
||||
# 没有配置模型,则用jieba
|
||||
if not knowledge_llm.qa_similar_model_id:
|
||||
return None
|
||||
return cls.get_bisheng_llm(model_id=knowledge_llm.qa_similar_model_id)
|
||||
|
||||
@classmethod
|
||||
def get_knowledge_default_embedding(cls) -> Optional[Embeddings]:
|
||||
""" 获取知识库默认的embedding模型 """
|
||||
knowledge_llm = cls.get_knowledge_llm()
|
||||
# 没有配置模型,则用jieba
|
||||
if not knowledge_llm.embedding_model_id:
|
||||
return None
|
||||
return cls.get_bisheng_embedding(model_id=knowledge_llm.embedding_model_id)
|
||||
|
||||
@classmethod
|
||||
def update_knowledge_llm(cls, request: Request, login_user: UserPayload, data: KnowledgeLLMConfig) \
|
||||
-> KnowledgeLLMConfig:
|
||||
""" 更新知识库相关的默认模型配置 """
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.KNOWLEDGE_LLM)
|
||||
if config:
|
||||
config.value = json.dumps(data.dict())
|
||||
else:
|
||||
config = Config(key=ConfigKeyEnum.KNOWLEDGE_LLM.value, value=json.dumps(data.dict()))
|
||||
ConfigDao.insert_config(config)
|
||||
return data
|
||||
|
||||
@classmethod
|
||||
def get_assistant_llm(cls) -> AssistantLLMConfig:
|
||||
""" 获取助手相关的默认模型配置 """
|
||||
ret = {}
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.ASSISTANT_LLM)
|
||||
if config:
|
||||
ret = json.loads(config.value)
|
||||
return AssistantLLMConfig(**ret)
|
||||
|
||||
@classmethod
|
||||
def update_assistant_llm(cls, request: Request, login_user: UserPayload, data: AssistantLLMConfig) \
|
||||
-> AssistantLLMConfig:
|
||||
""" 更新助手相关的默认模型配置 """
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.ASSISTANT_LLM)
|
||||
if config:
|
||||
config.value = json.dumps(data.dict())
|
||||
else:
|
||||
config = Config(key=ConfigKeyEnum.ASSISTANT_LLM.value, value=json.dumps(data.dict()))
|
||||
ConfigDao.insert_config(config)
|
||||
return data
|
||||
|
||||
@classmethod
|
||||
def get_evaluation_llm(cls) -> EvaluationLLMConfig:
|
||||
""" 获取评测功能的默认模型配置 """
|
||||
ret = {}
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.EVALUATION_LLM)
|
||||
if config:
|
||||
ret = json.loads(config.value)
|
||||
return EvaluationLLMConfig(**ret)
|
||||
|
||||
@classmethod
|
||||
def get_evaluation_llm_object(cls) -> BaseChatModel:
|
||||
evaluation_llm = cls.get_evaluation_llm()
|
||||
if not evaluation_llm.model_id:
|
||||
raise Exception('未配置评测模型')
|
||||
return cls.get_bisheng_llm(model_id=evaluation_llm.model_id)
|
||||
|
||||
@classmethod
|
||||
def get_bisheng_llm(cls, **kwargs) -> BaseChatModel:
|
||||
""" 获取评测功能的默认模型配置 """
|
||||
class_object = import_by_type(_type='llms', name='BishengLLM')
|
||||
return instantiate_llm('BishengLLM', class_object, kwargs)
|
||||
|
||||
@classmethod
|
||||
def get_bisheng_embedding(cls, **kwargs) -> Embeddings:
|
||||
""" 获取评测功能的默认模型配置 """
|
||||
class_object = import_by_type(_type='embeddings', name='BishengEmbedding')
|
||||
return instantiate_embedding(class_object, kwargs)
|
||||
|
||||
@classmethod
|
||||
def update_evaluation_llm(cls, request: Request, login_user: UserPayload, data: EvaluationLLMConfig) \
|
||||
-> EvaluationLLMConfig:
|
||||
""" 更新评测功能的默认模型配置 """
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.EVALUATION_LLM)
|
||||
if config:
|
||||
config.value = json.dumps(data.dict())
|
||||
else:
|
||||
config = Config(key=ConfigKeyEnum.EVALUATION_LLM.value, value=json.dumps(data.dict()))
|
||||
ConfigDao.insert_config(config)
|
||||
return data
|
||||
|
||||
@classmethod
|
||||
def update_workflow_llm(cls, request: Request, login_user: UserPayload, data: EvaluationLLMConfig) \
|
||||
-> EvaluationLLMConfig:
|
||||
""" 更新workflow的默认模型配置 """
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.WORKFLOW_LLM)
|
||||
if config:
|
||||
config.value = json.dumps(data.dict())
|
||||
else:
|
||||
config = Config(key=ConfigKeyEnum.WORKFLOW_LLM.value, value=json.dumps(data.dict()))
|
||||
ConfigDao.insert_config(config)
|
||||
return data
|
||||
|
||||
@classmethod
|
||||
def get_workflow_llm(cls) -> EvaluationLLMConfig:
|
||||
""" 获取评测功能的默认模型配置 """
|
||||
ret = {}
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.WORKFLOW_LLM)
|
||||
if config:
|
||||
ret = json.loads(config.value)
|
||||
return EvaluationLLMConfig(**ret)
|
||||
|
||||
@classmethod
|
||||
def get_assistant_llm_list(cls, request: Request, login_user: UserPayload) -> List[LLMServerInfo]:
|
||||
""" 获取助手可选的模型列表 """
|
||||
assistant_llm = cls.get_assistant_llm()
|
||||
if not assistant_llm.llm_list:
|
||||
return []
|
||||
model_list = LLMDao.get_model_by_ids([one.model_id for one in assistant_llm.llm_list])
|
||||
if not model_list:
|
||||
return []
|
||||
|
||||
default_llm = next(filter(lambda x: x.default, assistant_llm.llm_list), None)
|
||||
if not default_llm:
|
||||
default_llm = assistant_llm.llm_list[0]
|
||||
model_dict = {}
|
||||
default_server = None
|
||||
for one in model_list:
|
||||
if one.server_id not in model_dict:
|
||||
model_dict[one.server_id] = []
|
||||
if one.id == default_llm.model_id:
|
||||
default_server = one.server_id
|
||||
model_dict[one.server_id].insert(0, LLMModelInfo(**one.dict(exclude={'config'})))
|
||||
continue
|
||||
model_dict[one.server_id].append(LLMModelInfo(**one.dict(exclude={'config'})))
|
||||
server_list = LLMDao.get_server_by_ids(list(model_dict.keys()))
|
||||
|
||||
ret = []
|
||||
for one in server_list:
|
||||
if one.id == default_server:
|
||||
ret.insert(0, LLMServerInfo(**one.dict(exclude={'config'}), models=model_dict[one.id]))
|
||||
continue
|
||||
ret.append(LLMServerInfo(**one.dict(exclude={'config'}), models=model_dict[one.id]))
|
||||
|
||||
return ret
|
||||
|
||||
@classmethod
|
||||
async def update_workbench_llm(cls, config_obj: WorkbenchModelConfig, background_tasks: BackgroundTasks):
|
||||
"""
|
||||
更新灵思模型配置
|
||||
:param config_obj:
|
||||
:return:
|
||||
"""
|
||||
|
||||
config = await ConfigDao.aget_config(ConfigKeyEnum.LINSIGHT_LLM)
|
||||
if not config:
|
||||
config = Config(key=ConfigKeyEnum.LINSIGHT_LLM.value, value='{}')
|
||||
|
||||
if config_obj.embedding_model:
|
||||
# 判断是否一致
|
||||
config_old_obj = WorkbenchModelConfig(**json.loads(config.value)) if config else WorkbenchModelConfig()
|
||||
if (config_obj.embedding_model.id and config_old_obj.embedding_model is None or
|
||||
config_obj.embedding_model.id != config_old_obj.embedding_model.id):
|
||||
embeddings = decide_embeddings(config_obj.embedding_model.id)
|
||||
try:
|
||||
await embeddings.aembed_query("test")
|
||||
except Exception as e:
|
||||
raise Exception(f"Embedding模型初始化失败: {str(e)}")
|
||||
from bisheng.api.services.linsight.sop_manage import SOPManageService
|
||||
|
||||
background_tasks.add_task(SOPManageService.rebuild_sop_vector_store_task, embeddings)
|
||||
|
||||
config.value = json.dumps(config_obj.model_dump(), ensure_ascii=False)
|
||||
|
||||
await ConfigDao.async_insert_config(config)
|
||||
|
||||
return config_obj
|
||||
|
||||
@classmethod
|
||||
async def get_workbench_llm(cls) -> WorkbenchModelConfig:
|
||||
"""
|
||||
获取工作台模型配置
|
||||
:return:
|
||||
"""
|
||||
ret = {}
|
||||
config = await ConfigDao.aget_config(ConfigKeyEnum.LINSIGHT_LLM)
|
||||
if config:
|
||||
ret = json.loads(config.value)
|
||||
return WorkbenchModelConfig(**ret)
|
||||
@@ -0,0 +1,112 @@
|
||||
import pypandoc
|
||||
from loguru import logger
|
||||
from pathlib import Path
|
||||
from uuid import uuid4
|
||||
|
||||
try:
|
||||
# 尝试检查 pandoc 版本,如果失败则尝试下载
|
||||
pandoc_path = pypandoc.get_pandoc_path()
|
||||
logger.debug(f"Pandoc found at: {pandoc_path}")
|
||||
except OSError: # OSError 是 get_pandoc_path 在找不到时抛出的
|
||||
logger.debug("Pandoc not found. Attempting to download pandoc...")
|
||||
try:
|
||||
pypandoc.download_pandoc() # 这会下载到 pypandoc 的包目录中
|
||||
logger.debug("Pandoc downloaded successfully by pypandoc.")
|
||||
# 你可能需要重新获取路径或 pypandoc 之后会自动找到
|
||||
except Exception as e_download:
|
||||
logger.debug(f"Failed to download pandoc using pypandoc: {e_download}")
|
||||
exit() # 如果无法下载,则退出
|
||||
|
||||
|
||||
def convert_doc_to_md_pandoc_high_quality(
|
||||
doc_path_str: str, output_md_str: str, image_dir_name: str = "media"
|
||||
):
|
||||
"""
|
||||
使用 Pandoc 将 .doc 或 .docx 文件高质量地转换为 Markdown,并提取图片。
|
||||
|
||||
参数:
|
||||
doc_path_str (str): 输入的 Word 文档路径。
|
||||
output_md_str (str): 输出的 Markdown 文件路径。
|
||||
image_dir_name (str): 用于存放提取图片的子目录名称。此目录将创建在 Markdown 文件旁边。
|
||||
"""
|
||||
doc_path = Path(doc_path_str)
|
||||
output_md_path = Path(output_md_str)
|
||||
|
||||
if not doc_path.exists():
|
||||
logger.debug(f"错误:输入文件 {doc_path} 不存在。")
|
||||
return
|
||||
|
||||
# 确保输出 Markdown 文件的父目录存在
|
||||
output_md_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Pandoc 输出格式选项 (gfm 通常是好选择)
|
||||
pandoc_format_to = "gfm"
|
||||
|
||||
# Pandoc 额外参数
|
||||
# --extract-media=目录名: 告诉 Pandoc 提取所有媒体文件(如图片)到指定的子目录。
|
||||
# Pandoc 会自动创建此目录,并使 Markdown 中的图片链接指向此目录。
|
||||
# --atx-headers: 如果你的 Pandoc 版本支持,此选项会使用 '#' 样式的标题。
|
||||
# 如果之前因版本问题报错,而你没有升级 Pandoc,可以注释掉此行。
|
||||
extra_args = [
|
||||
"--wrap=none",
|
||||
# '--atx-headers', # 如果 Pandoc 版本较旧导致此选项报错,请注释掉或升级 Pandoc
|
||||
f"--extract-media={image_dir_name}", # 关键:提取图片到指定子目录
|
||||
]
|
||||
|
||||
# 图片将被提取到 output_md_path 同级目录下的 image_dir_name 子目录中
|
||||
# 例如:如果 output_md_path 是 "output/document.md" 且 image_dir_name 是 "images",
|
||||
# 图片将存放在 "output/images/" 目录下,链接会是 "images/image1.png"
|
||||
|
||||
try:
|
||||
pypandoc.convert_file(
|
||||
source_file=str(doc_path),
|
||||
to=pandoc_format_to,
|
||||
outputfile=str(output_md_path),
|
||||
extra_args=extra_args,
|
||||
)
|
||||
logger.debug(f"Pandoc 转换完成: {output_md_path}")
|
||||
|
||||
except RuntimeError as e: # Pandoc 未找到或执行错误时常抛出 RuntimeError
|
||||
if "Unknown option --atx-headers" in str(e):
|
||||
logger.debug(
|
||||
" 错误提示 '--atx-headers' 选项未知,这通常意味着您的 Pandoc 版本较旧。"
|
||||
)
|
||||
except Exception as e: # 其他潜在错误
|
||||
logger.debug(f"转换文件 {doc_path} 时发生未知错误: {e}")
|
||||
|
||||
|
||||
def handler(cache_dir, file_name):
|
||||
"""
|
||||
处理文件转换的主函数。
|
||||
|
||||
参数:
|
||||
file_name (str): 输入的 Word 文档路径。
|
||||
knowledge_id (str): 知识 ID,用于生成输出文件名。
|
||||
"""
|
||||
doc_id = str(uuid4())
|
||||
md_file_name = f"{cache_dir}/{doc_id}.md"
|
||||
local_image_dir = f"{cache_dir}/{doc_id}"
|
||||
convert_doc_to_md_pandoc_high_quality(
|
||||
doc_path_str=file_name,
|
||||
output_md_str=md_file_name,
|
||||
image_dir_name=local_image_dir,
|
||||
)
|
||||
return md_file_name, f"{local_image_dir}/media", doc_id
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# 定义测试参数
|
||||
test_cache_dir = "/Users/tju/Desktop"
|
||||
test_file_name = "/Users/tju/Resources/docs/docx/resume.docx"
|
||||
# test_file_name = "/Users/tju/Resources/docs/docx/2307.09288.docx"
|
||||
|
||||
# 调用 handler 函数进行测试
|
||||
md_file_name, image_dir, doc_id = handler(
|
||||
cache_dir=test_cache_dir,
|
||||
file_name=test_file_name,
|
||||
)
|
||||
|
||||
# 输出结果
|
||||
print(f"Markdown 文件路径: {md_file_name}")
|
||||
print(f"图片目录路径: {image_dir}")
|
||||
print(f"文档 ID: {doc_id}")
|
||||
@@ -0,0 +1,453 @@
|
||||
import math
|
||||
import os
|
||||
from typing import List
|
||||
from uuid import uuid4
|
||||
|
||||
import openpyxl
|
||||
import pandas as pd
|
||||
from loguru import logger
|
||||
|
||||
|
||||
def xls_to_xlsx(xls_path):
|
||||
if not xls_path.lower().endswith(".xls"):
|
||||
return None
|
||||
|
||||
if not os.path.exists(xls_path):
|
||||
return None
|
||||
|
||||
try:
|
||||
xls_file = pd.ExcelFile(xls_path)
|
||||
sheets_to_write = {}
|
||||
|
||||
# 2. 遍历所有工作表,检查是否为空,并将非空内容存入字典
|
||||
for sheet_name in xls_file.sheet_names:
|
||||
df = xls_file.parse(sheet_name)
|
||||
# df.empty 会判断 DataFrame 是否无数据(行数为0)
|
||||
if not df.empty:
|
||||
sheets_to_write[sheet_name] = df
|
||||
else:
|
||||
# 丢弃空工作表
|
||||
pass
|
||||
|
||||
# 3. 如果没有任何非空工作表,则不创建新文件
|
||||
if not sheets_to_write:
|
||||
return None
|
||||
|
||||
# 4. 如果存在非空工作表,则写入新文件
|
||||
xlsx_path = os.path.splitext(xls_path)[0] + ".xlsx"
|
||||
with pd.ExcelWriter(xlsx_path, engine="openpyxl") as writer:
|
||||
for sheet_name, df in sheets_to_write.items():
|
||||
df.to_excel(writer, sheet_name=sheet_name, index=False)
|
||||
|
||||
return xlsx_path
|
||||
|
||||
except Exception as e:
|
||||
return None
|
||||
|
||||
|
||||
def remove_characters(s, chars_to_remove=["\n", "\r"]):
|
||||
"""
|
||||
从字符串中移除指定的字符。
|
||||
"""
|
||||
if not isinstance(s, str):
|
||||
return s
|
||||
for char in chars_to_remove:
|
||||
s = s.replace(char, "")
|
||||
return s.strip()
|
||||
|
||||
|
||||
def unmerge_and_read_sheet(sheet_obj):
|
||||
"""
|
||||
读取 openpyxl 工作表对象,通过将合并区域左上角的值填充到该区域的所有单元格中来取消合并单元格,
|
||||
并以列表的列表形式返回数据。
|
||||
"""
|
||||
if sheet_obj.max_row == 0 or sheet_obj.max_column == 0:
|
||||
return []
|
||||
max_row = sheet_obj.max_row
|
||||
max_column = sheet_obj.max_column
|
||||
data_grid = [
|
||||
[None for _ in range(max_column)] for _ in range(max_row)
|
||||
]
|
||||
|
||||
# 连续50行空行停止读取内容
|
||||
empty_row_num = 0
|
||||
max_empty_rows = 50
|
||||
empty_row_end = 0
|
||||
for r_idx, row in enumerate(sheet_obj.iter_rows()):
|
||||
if empty_row_num > max_empty_rows:
|
||||
break
|
||||
row_empty = True
|
||||
for c_idx, cell in enumerate(row):
|
||||
data_grid[r_idx][c_idx] = cell.value
|
||||
if cell.value:
|
||||
row_empty = False
|
||||
if row_empty:
|
||||
empty_row_num += 1
|
||||
empty_row_end = r_idx
|
||||
else:
|
||||
empty_row_num = 0
|
||||
empty_row_end = 0
|
||||
|
||||
merged_cell_ranges = list(sheet_obj.merged_cells.ranges)
|
||||
for merged_range in merged_cell_ranges:
|
||||
min_col, min_row, max_col, max_row = merged_range.bounds
|
||||
top_left_cell_value = sheet_obj.cell(row=min_row, column=min_col).value
|
||||
for r in range(min_row, max_row + 1):
|
||||
for c in range(min_col, max_col + 1):
|
||||
data_grid[r - 1][c - 1] = top_left_cell_value
|
||||
if empty_row_end and empty_row_end - max_empty_rows > 0:
|
||||
data_grid = data_grid[:empty_row_end - max_empty_rows]
|
||||
return data_grid
|
||||
|
||||
|
||||
def generate_markdown_table_string(
|
||||
header_rows_list_of_lists,
|
||||
data_rows_list_of_lists,
|
||||
num_columns,
|
||||
separator_placement_index=1,
|
||||
):
|
||||
"""
|
||||
根据新规则生成Markdown表格字符串。
|
||||
如果header_rows_list_of_lists为空,则不生成表头和分隔符。
|
||||
"""
|
||||
md_lines = []
|
||||
|
||||
# 只有在提供了表头行时,才处理表头和分隔符
|
||||
if header_rows_list_of_lists:
|
||||
pre_separator_header = header_rows_list_of_lists[:separator_placement_index]
|
||||
for row_values in pre_separator_header:
|
||||
md_lines.append(
|
||||
"| "
|
||||
+ " | ".join(
|
||||
remove_characters(str(v)) if v is not None else ""
|
||||
for v in row_values
|
||||
)
|
||||
+ " |"
|
||||
)
|
||||
|
||||
# 在第一行表头下方插入Markdown分隔符
|
||||
if num_columns > 0:
|
||||
md_lines.append("|" + "---|" * num_columns)
|
||||
|
||||
post_separator_header = header_rows_list_of_lists[separator_placement_index:]
|
||||
for row_values in post_separator_header:
|
||||
md_lines.append(
|
||||
"| "
|
||||
+ " | ".join(
|
||||
remove_characters(str(v)) if v is not None else ""
|
||||
for v in row_values
|
||||
)
|
||||
+ " |"
|
||||
)
|
||||
|
||||
# 总是处理数据行
|
||||
for row_values in data_rows_list_of_lists:
|
||||
md_lines.append(
|
||||
"| "
|
||||
+ " | ".join(
|
||||
remove_characters(str(v)) if v is not None else "" for v in row_values
|
||||
)
|
||||
+ " |"
|
||||
)
|
||||
|
||||
return "\n".join(md_lines)
|
||||
|
||||
|
||||
def process_dataframe_to_markdown_files(
|
||||
df,
|
||||
sheet_index: str,
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
output_dir,
|
||||
append_header=True,
|
||||
):
|
||||
"""
|
||||
- append_header=True: 按 num_header_rows 分离表头和数据。
|
||||
- append_header=False: 全部内容视为数据,表头为空,忽略 num_header_rows。
|
||||
"""
|
||||
if df.empty:
|
||||
logger.warning(f" 源 '{sheet_index}' 的数据DataFrame为空,跳过Markdown生成。")
|
||||
return
|
||||
|
||||
num_columns = df.shape[1]
|
||||
rows = df.shape[0]
|
||||
|
||||
if rows == 0 or num_columns == 0:
|
||||
return
|
||||
|
||||
header_block_df = pd.DataFrame()
|
||||
start_header_idx, end_header_idx = num_header_rows[0], num_header_rows[1]
|
||||
if start_header_idx >= rows:
|
||||
append_header = False
|
||||
|
||||
# --- 核心逻辑修改:根据 append_header 决定如何切分数据 ---
|
||||
if append_header:
|
||||
# 根据用户规则处理表头索引越界问题
|
||||
if start_header_idx >= rows:
|
||||
logger.warning(f" 表头起始行 {start_header_idx} 超出总行数 {rows}。将使用第一行作为表头。")
|
||||
start_header_idx, end_header_idx = 0, 0
|
||||
elif end_header_idx >= rows:
|
||||
logger.warning(f" 表头结束行 {end_header_idx} 超出总行数 {rows}。将截断至最后一行。")
|
||||
end_header_idx = rows - 1
|
||||
|
||||
# 确保索引合法
|
||||
if start_header_idx < 0: start_header_idx = 0
|
||||
if end_header_idx < start_header_idx: end_header_idx = start_header_idx
|
||||
|
||||
try:
|
||||
header_slice = slice(start_header_idx, end_header_idx + 1)
|
||||
header_block_df = df.iloc[header_slice]
|
||||
data_block_df = df.drop(df.index[header_slice]).reset_index(drop=True)
|
||||
header_rows_as_lists = header_block_df.values.tolist()
|
||||
except Exception as e:
|
||||
logger.error(
|
||||
f" 在源 '{sheet_index}' 中根据表头索引 [{start_header_idx}, {end_header_idx}] 切分数据时出错: {e}。跳过。")
|
||||
return
|
||||
else:
|
||||
# 当 append_header 为 False 时,所有内容都视为数据,表头列表为空
|
||||
header_rows_as_lists = []
|
||||
data_block_df = df.reset_index(drop=True)
|
||||
|
||||
# --- 后续分页逻辑 ---
|
||||
if data_block_df.empty:
|
||||
if append_header and not header_block_df.empty:
|
||||
markdown_content = generate_markdown_table_string(
|
||||
header_rows_as_lists, [], num_columns
|
||||
)
|
||||
# BUG FIX: Use zfill for proper padding. This is file '000' for the sheet.
|
||||
file_name = f"{str(sheet_index).zfill(2)}000.md"
|
||||
file_path = os.path.join(output_dir, file_name)
|
||||
try:
|
||||
with open(file_path, "w", encoding="utf-8") as f:
|
||||
f.write(markdown_content)
|
||||
logger.debug(f" 已保存仅含表头的文件:'{file_path}'")
|
||||
except Exception as e:
|
||||
logger.debug(f" 保存文件 '{file_path}' 时出错: {e}")
|
||||
return
|
||||
|
||||
num_data_rows_total = len(data_block_df)
|
||||
num_files_to_create = math.ceil(num_data_rows_total / rows_per_markdown) if rows_per_markdown > 0 else (
|
||||
1 if num_data_rows_total > 0 else 0)
|
||||
|
||||
for i in range(num_files_to_create):
|
||||
start_idx = i * rows_per_markdown
|
||||
end_idx = min(start_idx + rows_per_markdown, num_data_rows_total)
|
||||
current_data_chunk_as_lists = data_block_df.iloc[start_idx:end_idx].values.tolist()
|
||||
|
||||
final_header_for_chunk = header_rows_as_lists
|
||||
final_data_for_chunk = current_data_chunk_as_lists
|
||||
|
||||
# 如果不附加真实表头,并且当前数据块不为空,则将数据的第一行用作“伪表头”以生成分隔符
|
||||
if not append_header and current_data_chunk_as_lists:
|
||||
final_header_for_chunk = [current_data_chunk_as_lists[0]]
|
||||
final_data_for_chunk = current_data_chunk_as_lists[1:]
|
||||
|
||||
markdown_content = generate_markdown_table_string(
|
||||
final_header_for_chunk, final_data_for_chunk, num_columns
|
||||
)
|
||||
|
||||
# BUG FIX: Use zfill for proper 2-digit sheet and 3-digit file padding.
|
||||
file_name = f"{str(sheet_index).zfill(2)}{str(i).zfill(3)}.md"
|
||||
file_path = os.path.join(output_dir, file_name)
|
||||
|
||||
try:
|
||||
with open(file_path, "w", encoding="utf-8") as f:
|
||||
f.write(markdown_content)
|
||||
logger.debug(
|
||||
f" 已保存:'{file_path}' (含 {len(current_data_chunk_as_lists)} 行原始数据)"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(f" 保存文件 '{file_path}' 时出错: {e}")
|
||||
|
||||
|
||||
def is_list_of_lists_empty(data_list):
|
||||
"""
|
||||
判断一个二维列表是否为空或只包含空值 (None, '')。
|
||||
"""
|
||||
if not data_list:
|
||||
return True
|
||||
# 使用 any() 和生成器表达式,高效判断
|
||||
# any(row) 检查是否存在非空行
|
||||
# any(cell is not None and cell != '' for cell in row) 检查行内是否有非空单元格
|
||||
return not any(any(cell is not None and str(cell).strip() != '' for cell in row) for row in data_list)
|
||||
|
||||
|
||||
def excel_file_to_markdown(
|
||||
excel_path, num_header_rows, rows_per_markdown, output_dir, append_header=True
|
||||
):
|
||||
logger.debug(f"\n开始处理Excel文件:'{excel_path}'")
|
||||
try:
|
||||
workbook = openpyxl.load_workbook(excel_path, data_only=True, read_only=False)
|
||||
except Exception as e:
|
||||
logger.debug(f"错误:无法加载Excel文件 '{excel_path}'。原因: {e}")
|
||||
return
|
||||
|
||||
sheet_index = 0
|
||||
for sheet_name in workbook.sheetnames:
|
||||
logger.debug(f"\n 正在处理Excel工作表:'{sheet_name}'...")
|
||||
sheet_obj = workbook[sheet_name]
|
||||
unmerged_data_list_of_lists = unmerge_and_read_sheet(sheet_obj)
|
||||
logger.debug(f"\n <read all data>Excel<UNK>'{sheet_name}'...{len(unmerged_data_list_of_lists)}")
|
||||
|
||||
# 使用新的判断函数
|
||||
if is_list_of_lists_empty(unmerged_data_list_of_lists):
|
||||
logger.debug(f" 工作表 '{sheet_name}' 为空或无有效数据,跳过。")
|
||||
continue
|
||||
|
||||
df = pd.DataFrame(unmerged_data_list_of_lists)
|
||||
df.fillna("", inplace=True)
|
||||
if df.empty:
|
||||
logger.debug(f" 工作表 '{sheet_name}' 处理后为空DataFrame,跳过。")
|
||||
continue
|
||||
|
||||
process_dataframe_to_markdown_files(
|
||||
df,
|
||||
str(sheet_index),
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
output_dir,
|
||||
append_header=append_header,
|
||||
)
|
||||
sheet_index += 1
|
||||
|
||||
if workbook:
|
||||
workbook.close()
|
||||
logger.debug(f"\nExcel文件 '{excel_path}' 处理完成。")
|
||||
|
||||
|
||||
def csv_file_to_markdown(
|
||||
csv_path,
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
output_dir,
|
||||
csv_encoding="utf-8",
|
||||
csv_delimiter=",",
|
||||
append_header=True,
|
||||
):
|
||||
logger.debug(f"\n开始处理CSV文件:'{csv_path}'")
|
||||
try:
|
||||
df = pd.read_csv(
|
||||
csv_path,
|
||||
header=None,
|
||||
dtype=str,
|
||||
encoding=csv_encoding,
|
||||
sep=csv_delimiter,
|
||||
keep_default_na=False,
|
||||
)
|
||||
df.fillna("", inplace=True)
|
||||
|
||||
except pd.errors.EmptyDataError:
|
||||
logger.debug(f"错误:CSV文件 '{csv_path}' 为空。")
|
||||
return
|
||||
except FileNotFoundError:
|
||||
logger.debug(f"错误:CSV文件 '{csv_path}' 未找到。")
|
||||
return
|
||||
except Exception as e:
|
||||
logger.debug(f"错误:无法读取CSV文件 '{csv_path}'。原因: {e}")
|
||||
return
|
||||
|
||||
if df.empty:
|
||||
logger.debug(f"CSV文件 '{csv_path}' 为空或处理后为空,跳过。")
|
||||
return
|
||||
|
||||
process_dataframe_to_markdown_files(
|
||||
df,
|
||||
"0",
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
output_dir,
|
||||
append_header,
|
||||
)
|
||||
logger.debug(f"\nCSV文件 '{csv_path}' 处理完成。")
|
||||
|
||||
|
||||
def convert_file_to_markdown(
|
||||
input_file_path,
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
base_output_dir="output_markdown_files",
|
||||
csv_encoding="utf-8",
|
||||
csv_delimiter=",",
|
||||
append_header=True,
|
||||
):
|
||||
"""
|
||||
将 Excel 或 CSV 文件转换为多个 Markdown 文件。
|
||||
"""
|
||||
if not os.path.exists(input_file_path):
|
||||
logger.debug(f"错误:输入文件 '{input_file_path}' 未找到。")
|
||||
return
|
||||
|
||||
if not os.path.exists(base_output_dir):
|
||||
os.makedirs(base_output_dir)
|
||||
logger.debug(f"创建输出目录:'{base_output_dir}'")
|
||||
|
||||
_, file_extension = os.path.splitext(input_file_path)
|
||||
file_extension = file_extension.lower()
|
||||
if file_extension == ".xls":
|
||||
input_file_path = xls_to_xlsx(input_file_path)
|
||||
|
||||
if file_extension in [".xlsx", ".xls"]:
|
||||
excel_file_to_markdown(
|
||||
input_file_path,
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
base_output_dir,
|
||||
append_header,
|
||||
)
|
||||
elif file_extension == ".csv":
|
||||
csv_file_to_markdown(
|
||||
input_file_path,
|
||||
num_header_rows,
|
||||
rows_per_markdown,
|
||||
base_output_dir,
|
||||
csv_encoding,
|
||||
csv_delimiter,
|
||||
append_header,
|
||||
)
|
||||
else:
|
||||
logger.debug(
|
||||
f"错误:不支持的文件类型 '{file_extension}'。请提供 Excel (.xlsx, .xls) 或 CSV (.csv) 文件。"
|
||||
)
|
||||
|
||||
|
||||
def handler(
|
||||
cache_dir,
|
||||
file_name: str,
|
||||
header_rows: List[int] = [0, 1],
|
||||
data_rows: int = 12,
|
||||
append_header=True,
|
||||
):
|
||||
"""
|
||||
处理文件转换的主函数。
|
||||
"""
|
||||
doc_id = uuid4()
|
||||
md_file_name = f"{cache_dir}/{doc_id}"
|
||||
|
||||
convert_file_to_markdown(
|
||||
input_file_path=file_name,
|
||||
base_output_dir=md_file_name,
|
||||
num_header_rows=header_rows,
|
||||
rows_per_markdown=data_rows,
|
||||
append_header=append_header,
|
||||
)
|
||||
return md_file_name, None, doc_id
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# 定义测试参数
|
||||
test_cache_dir = "/Users/zhangguoqing/Downloads/tmp"
|
||||
test_file_name = "/Users/zhangguoqing/Downloads/124327.xlsx"
|
||||
# 测试 append_header=True 且索引越界的情况
|
||||
test_header_rows = [0, 0] # start_header_index 超出范围
|
||||
test_data_rows = 2
|
||||
test_append_header = True
|
||||
|
||||
# 调用 handler 函数
|
||||
print("--- 测试场景: append_header=True, 表头索引越界 ---")
|
||||
handler(
|
||||
cache_dir=test_cache_dir,
|
||||
file_name=test_file_name,
|
||||
header_rows=test_header_rows,
|
||||
data_rows=test_data_rows,
|
||||
append_header=test_append_header,
|
||||
)
|
||||
@@ -0,0 +1,742 @@
|
||||
import requests
|
||||
from bs4 import BeautifulSoup, Comment
|
||||
from markdownify import markdownify as md
|
||||
import os
|
||||
import re
|
||||
import base64
|
||||
from urllib.parse import urljoin, urlparse
|
||||
from uuid import uuid4
|
||||
from loguru import logger
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
# Configure logger
|
||||
|
||||
USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36"
|
||||
|
||||
|
||||
class HTML2MarkdownConverter:
|
||||
def __init__(
|
||||
self,
|
||||
output_dir="output",
|
||||
media_download_timeout=60,
|
||||
):
|
||||
self.output_dir = output_dir
|
||||
self.MEDIA_DOWNLOAD_TIMEOUT = media_download_timeout
|
||||
os.makedirs(self.output_dir, exist_ok=True)
|
||||
|
||||
self.current_image_absolute_path = None
|
||||
self.current_video_absolute_path = None
|
||||
self.base_url = None
|
||||
# mhtml_resources is kept for cid processing in case it's used by other parts, but parsing is removed.
|
||||
self.mhtml_resources = {}
|
||||
self.source_html_filepath = None
|
||||
|
||||
def _clean_html(self, html_content):
|
||||
logger.debug("Starting HTML cleaning (refined logic).")
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
for D_tag in soup.find_all(["script", "style", "link", "meta"]):
|
||||
D_tag.decompose()
|
||||
for comment in soup.find_all(string=lambda text: isinstance(text, Comment)):
|
||||
comment.extract()
|
||||
potentially_problematic_container_tags = ["header", "footer", "nav", "aside"]
|
||||
non_content_patterns = re.compile(
|
||||
r"adsbygoogle|ad-slot|advertisement|promo(tion)?|banner-ad|popup-ad|cookie-notice|gdpr-banner|newsletter-signup|social-share-buttons|flyout-menu",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
non_content_roles = [
|
||||
"banner",
|
||||
"navigation",
|
||||
"search",
|
||||
"complementary",
|
||||
"contentinfo",
|
||||
"dialog",
|
||||
"menubar",
|
||||
"toolbar",
|
||||
"directory",
|
||||
"log",
|
||||
"status",
|
||||
"timer",
|
||||
]
|
||||
media_tags_to_check = ["img", "video", "picture", "figure", "svg", "audio"]
|
||||
for tag in list(soup.find_all(True)):
|
||||
if not tag.parent:
|
||||
continue
|
||||
decomposed_this_iteration = False
|
||||
if tag.name in potentially_problematic_container_tags:
|
||||
if not tag.find_all(media_tags_to_check):
|
||||
tag.decompose()
|
||||
decomposed_this_iteration = True
|
||||
if decomposed_this_iteration:
|
||||
continue
|
||||
if tag.name not in media_tags_to_check:
|
||||
class_match = any(
|
||||
non_content_patterns.search(cls) for cls in tag.get("class", [])
|
||||
)
|
||||
id_match = (
|
||||
non_content_patterns.search(tag.get("id", ""))
|
||||
if tag.get("id")
|
||||
else False
|
||||
)
|
||||
role_match = tag.get("role", "") in non_content_roles
|
||||
if class_match or id_match or role_match:
|
||||
if not tag.find_all(media_tags_to_check):
|
||||
tag.decompose()
|
||||
decomposed_this_iteration = True
|
||||
if decomposed_this_iteration:
|
||||
continue
|
||||
form_elements_to_remove = [
|
||||
"form",
|
||||
"button",
|
||||
"input",
|
||||
"select",
|
||||
"textarea",
|
||||
"fieldset",
|
||||
"legend",
|
||||
]
|
||||
for tag_name_to_remove in form_elements_to_remove:
|
||||
for form_tag in list(soup.find_all(tag_name_to_remove)):
|
||||
if not form_tag.parent:
|
||||
continue
|
||||
if not form_tag.find_all(media_tags_to_check):
|
||||
form_tag.decompose()
|
||||
for tag in soup.find_all(True):
|
||||
if not tag.parent and tag.name not in ["html", "head", "body"]:
|
||||
continue
|
||||
attrs_to_remove = [
|
||||
attr for attr in tag.attrs if attr.startswith("on") or attr == "style"
|
||||
]
|
||||
for attr in attrs_to_remove:
|
||||
del tag.attrs[attr]
|
||||
logger.debug("HTML cleaning (refined logic) finished.")
|
||||
return str(soup)
|
||||
|
||||
def _download_media_file(
|
||||
self,
|
||||
media_url,
|
||||
base_url_for_relative,
|
||||
media_absolute_save_dir,
|
||||
markdown_relative_media_folder,
|
||||
media_type_prefixes=("image/", "video/", "audio/"),
|
||||
):
|
||||
if not media_absolute_save_dir:
|
||||
logger.error(
|
||||
f"Absolute path for saving media (media_absolute_save_dir) is not set for URL: {media_url}"
|
||||
)
|
||||
return None, media_url
|
||||
original_media_url_for_error_logger = media_url
|
||||
try:
|
||||
parsed_media_url = urlparse(media_url)
|
||||
if media_url.startswith("data:"):
|
||||
if not any(
|
||||
prefix in media_url
|
||||
for prefix in media_type_prefixes
|
||||
if prefix == "image/"
|
||||
):
|
||||
return None, media_url
|
||||
try:
|
||||
header, encoded = media_url.split(",", 1)
|
||||
media_data = base64.b64decode(encoded)
|
||||
ext_match = re.search(
|
||||
r"data:(?P<type>image|video|audio)/(?P<ext>[a-zA-Z0-9+]+);",
|
||||
header,
|
||||
)
|
||||
ext = ext_match.group("ext").lower() if ext_match else "png"
|
||||
if ext == "svg+xml":
|
||||
ext = "svg"
|
||||
elif ext == "jpeg":
|
||||
ext = "jpg"
|
||||
|
||||
if not ext or len(ext) > 5 or not ext.isalnum():
|
||||
ext = "png"
|
||||
unique_filename = f"media_{uuid4().hex}.{ext}"
|
||||
absolute_filepath = os.path.join(
|
||||
media_absolute_save_dir, unique_filename
|
||||
)
|
||||
markdown_path = os.path.join(
|
||||
markdown_relative_media_folder, unique_filename
|
||||
)
|
||||
with open(absolute_filepath, "wb") as f:
|
||||
f.write(media_data)
|
||||
logger.info(f"Data URI media saved to {absolute_filepath}")
|
||||
return markdown_path, media_url
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to decode/save data URI media: {e}")
|
||||
return None, media_url
|
||||
|
||||
actual_media_url_str = media_url
|
||||
if not parsed_media_url.scheme or not parsed_media_url.netloc:
|
||||
if not base_url_for_relative:
|
||||
logger.warning(
|
||||
f"Cannot resolve relative media URL {actual_media_url_str} without a base URL."
|
||||
)
|
||||
return None, actual_media_url_str
|
||||
actual_media_url_str = urljoin(
|
||||
base_url_for_relative, actual_media_url_str
|
||||
)
|
||||
|
||||
parsed_actual_url = urlparse(actual_media_url_str)
|
||||
|
||||
ext = None
|
||||
path_part_for_ext = parsed_actual_url.path
|
||||
filename_from_url_for_ext = os.path.basename(path_part_for_ext)
|
||||
if "." in filename_from_url_for_ext:
|
||||
candidate_ext = filename_from_url_for_ext.split(".")[-1].lower()
|
||||
if (
|
||||
len(candidate_ext) <= 5
|
||||
and candidate_ext.isalnum()
|
||||
and candidate_ext
|
||||
in [
|
||||
"jpg",
|
||||
"jpeg",
|
||||
"png",
|
||||
"gif",
|
||||
"svg",
|
||||
"webp",
|
||||
"bmp",
|
||||
"tiff",
|
||||
"mp4",
|
||||
"webm",
|
||||
"ogg",
|
||||
"mov",
|
||||
"avi",
|
||||
"mkv",
|
||||
"mp3",
|
||||
"wav",
|
||||
"aac",
|
||||
]
|
||||
):
|
||||
ext = candidate_ext
|
||||
|
||||
if parsed_actual_url.scheme == "file":
|
||||
local_file_path_str = parsed_actual_url.path
|
||||
if (
|
||||
os.name == "nt"
|
||||
): # Windows: remove leading '/' if path starts like /C:/...
|
||||
if (
|
||||
len(local_file_path_str) > 2
|
||||
and local_file_path_str[0] == "/"
|
||||
and local_file_path_str[2] == ":"
|
||||
):
|
||||
local_file_path_str = local_file_path_str[1:]
|
||||
|
||||
local_file_to_copy = Path(local_file_path_str)
|
||||
|
||||
if local_file_to_copy.exists() and local_file_to_copy.is_file():
|
||||
if not ext:
|
||||
ext = (
|
||||
local_file_to_copy.suffix[1:].lower() or "dat"
|
||||
) # Get ext from local file if not from URL
|
||||
unique_filename = f"media_{uuid4().hex}.{ext}"
|
||||
absolute_filepath_dest = os.path.join(
|
||||
media_absolute_save_dir, unique_filename
|
||||
)
|
||||
markdown_path = os.path.join(
|
||||
markdown_relative_media_folder, unique_filename
|
||||
)
|
||||
shutil.copy(str(local_file_to_copy), absolute_filepath_dest)
|
||||
logger.info(
|
||||
f"Local media file '{local_file_to_copy}' copied to '{absolute_filepath_dest}'"
|
||||
)
|
||||
return (
|
||||
markdown_path,
|
||||
media_url,
|
||||
) # Return original media_url for consistency
|
||||
else:
|
||||
logger.warning(
|
||||
f"Local file '{local_file_to_copy}' referenced by '{actual_media_url_str}' not found or not a file."
|
||||
)
|
||||
return None, media_url
|
||||
|
||||
elif parsed_actual_url.scheme in ["http", "https"]:
|
||||
response = requests.get(
|
||||
actual_media_url_str,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
timeout=self.MEDIA_DOWNLOAD_TIMEOUT,
|
||||
stream=True,
|
||||
)
|
||||
response.raise_for_status()
|
||||
content_type = response.headers.get("Content-Type", "").lower()
|
||||
if not ext: # Try to get extension from Content-Type if not from URL
|
||||
if any(
|
||||
content_type.startswith(prefix)
|
||||
for prefix in media_type_prefixes
|
||||
):
|
||||
type_part = content_type.split(";")[0]
|
||||
candidate_ext_ct = type_part.split("/")[-1]
|
||||
if candidate_ext_ct == "svg+xml":
|
||||
ext = "svg"
|
||||
elif candidate_ext_ct == "jpeg":
|
||||
ext = "jpg"
|
||||
elif candidate_ext_ct in [
|
||||
"png",
|
||||
"gif",
|
||||
"webp",
|
||||
"bmp",
|
||||
"tiff",
|
||||
"mp4",
|
||||
"webm",
|
||||
"ogg",
|
||||
"mov",
|
||||
"avi",
|
||||
"mkv",
|
||||
"mp3",
|
||||
"wav",
|
||||
"aac",
|
||||
]:
|
||||
ext = candidate_ext_ct
|
||||
|
||||
ext = ext if ext else "dat" # Final fallback extension
|
||||
unique_filename = f"media_{uuid4().hex}.{ext}"
|
||||
absolute_filepath = os.path.join(
|
||||
media_absolute_save_dir, unique_filename
|
||||
)
|
||||
markdown_path = os.path.join(
|
||||
markdown_relative_media_folder, unique_filename
|
||||
)
|
||||
with open(absolute_filepath, "wb") as f:
|
||||
for chunk in response.iter_content(chunk_size=81920):
|
||||
f.write(chunk)
|
||||
logger.info(
|
||||
f"HTTP/S media {actual_media_url_str} downloaded to {absolute_filepath}"
|
||||
)
|
||||
return (
|
||||
markdown_path,
|
||||
actual_media_url_str,
|
||||
) # Return resolved URL for HTTP/S
|
||||
else:
|
||||
logger.warning(
|
||||
f"Skipping download for unsupported scheme: {actual_media_url_str}"
|
||||
)
|
||||
return None, actual_media_url_str
|
||||
|
||||
except requests.exceptions.Timeout:
|
||||
logger.error(
|
||||
f"Timeout processing media {original_media_url_for_error_logger}"
|
||||
)
|
||||
except requests.exceptions.HTTPError as e:
|
||||
logger.error(
|
||||
f"HTTP error {e.response.status_code} processing media {original_media_url_for_error_logger}: {e.response.reason}"
|
||||
)
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.error(
|
||||
f"RequestException processing media {original_media_url_for_error_logger}: {e}"
|
||||
)
|
||||
except IOError as e:
|
||||
logger.error(
|
||||
f"IOError processing media {original_media_url_for_error_logger}: {e}"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(
|
||||
f"Unexpected error processing media {original_media_url_for_error_logger}: {e}"
|
||||
)
|
||||
return None, original_media_url_for_error_logger
|
||||
|
||||
def _process_images_in_html(
|
||||
self, html_content, base_url_for_relative, markdown_relative_image_folder
|
||||
):
|
||||
logger.debug(
|
||||
f"Starting image processing. MD relative image folder: {markdown_relative_image_folder}"
|
||||
)
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
for img_tag in soup.find_all("img"):
|
||||
original_src = img_tag.get("src")
|
||||
alt_text = img_tag.get("alt", "").strip()
|
||||
if not original_src:
|
||||
img_tag.decompose()
|
||||
continue
|
||||
original_src = original_src.strip()
|
||||
if not original_src:
|
||||
img_tag.decompose()
|
||||
continue
|
||||
if original_src.startswith("cid:"):
|
||||
cid = original_src[4:]
|
||||
if hasattr(self, "mhtml_resources") and cid in self.mhtml_resources:
|
||||
media_data, resource_filename_ext = self.mhtml_resources[cid]
|
||||
ext_from_mhtml = "png"
|
||||
if "." in resource_filename_ext:
|
||||
candidate_ext = resource_filename_ext.split(".")[-1].lower()
|
||||
if len(candidate_ext) <= 5 and candidate_ext.isalnum():
|
||||
ext_from_mhtml = candidate_ext
|
||||
unique_filename = f"image_{uuid4().hex}.{ext_from_mhtml}"
|
||||
absolute_filepath = os.path.join(
|
||||
self.current_image_absolute_path, unique_filename
|
||||
)
|
||||
markdown_path = os.path.join(
|
||||
markdown_relative_image_folder, unique_filename
|
||||
)
|
||||
try:
|
||||
with open(absolute_filepath, "wb") as f:
|
||||
f.write(media_data)
|
||||
img_tag["src"] = markdown_path
|
||||
if not alt_text:
|
||||
alt_text = f"Embedded image {unique_filename}"
|
||||
img_tag["alt"] = alt_text
|
||||
except IOError as e:
|
||||
img_tag.decompose()
|
||||
else:
|
||||
img_tag.decompose()
|
||||
continue
|
||||
markdown_image_path, _ = self._download_media_file(
|
||||
original_src,
|
||||
base_url_for_relative,
|
||||
self.current_image_absolute_path,
|
||||
markdown_relative_image_folder,
|
||||
media_type_prefixes=("image/",),
|
||||
)
|
||||
if markdown_image_path:
|
||||
img_tag["src"] = markdown_image_path
|
||||
if not alt_text:
|
||||
alt_text = (
|
||||
f"Downloaded image {os.path.basename(markdown_image_path)}"
|
||||
)
|
||||
img_tag["alt"] = alt_text
|
||||
else:
|
||||
img_tag.decompose()
|
||||
logger.debug("Image processing finished.")
|
||||
return str(soup)
|
||||
|
||||
def _process_videos_in_html(
|
||||
self,
|
||||
html_content,
|
||||
base_url_for_relative,
|
||||
markdown_relative_video_folder,
|
||||
markdown_relative_image_folder_for_poster,
|
||||
):
|
||||
logger.debug(
|
||||
f"Starting video processing. MD video folder: {markdown_relative_video_folder}, MD poster folder: {markdown_relative_image_folder_for_poster}"
|
||||
)
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
for video_tag in soup.find_all("video"):
|
||||
original_poster_src = video_tag.get("poster")
|
||||
if original_poster_src:
|
||||
original_poster_src = original_poster_src.strip()
|
||||
if original_poster_src:
|
||||
logger.info(f"Processing poster for video: {original_poster_src}")
|
||||
if (
|
||||
self.current_image_absolute_path
|
||||
): # Ensure image path is set for saving posters
|
||||
poster_md_path, _ = self._download_media_file(
|
||||
original_poster_src,
|
||||
base_url_for_relative,
|
||||
self.current_image_absolute_path, # Save posters in the image asset directory
|
||||
markdown_relative_image_folder_for_poster, # Use the image folder's relative path for MD link
|
||||
media_type_prefixes=("image/",),
|
||||
)
|
||||
if poster_md_path:
|
||||
video_tag["poster"] = poster_md_path
|
||||
else:
|
||||
if "poster" in video_tag.attrs:
|
||||
del video_tag["poster"]
|
||||
else:
|
||||
logger.warning(
|
||||
f"Cannot process poster {original_poster_src} as image asset path is not initialized."
|
||||
)
|
||||
source_tags = video_tag.find_all("source")
|
||||
processed_source_successfully = False
|
||||
if source_tags:
|
||||
for source_tag in source_tags:
|
||||
original_src = source_tag.get("src")
|
||||
if original_src:
|
||||
original_src = original_src.strip()
|
||||
if not original_src:
|
||||
continue
|
||||
if original_src.startswith("cid:"):
|
||||
cid = original_src[4:]
|
||||
if (
|
||||
hasattr(self, "mhtml_resources")
|
||||
and cid in self.mhtml_resources
|
||||
):
|
||||
media_data, resource_filename_ext = (
|
||||
self.mhtml_resources[cid]
|
||||
)
|
||||
ext_from_mhtml = "mp4"
|
||||
if "." in resource_filename_ext:
|
||||
candidate_ext = resource_filename_ext.split(".")[
|
||||
-1
|
||||
].lower()
|
||||
if (
|
||||
len(candidate_ext) <= 5
|
||||
and candidate_ext.isalnum()
|
||||
):
|
||||
ext_from_mhtml = candidate_ext
|
||||
unique_filename = (
|
||||
f"video_{uuid4().hex}.{ext_from_mhtml}"
|
||||
)
|
||||
absolute_filepath = os.path.join(
|
||||
self.current_video_absolute_path, unique_filename
|
||||
)
|
||||
markdown_path = os.path.join(
|
||||
markdown_relative_video_folder, unique_filename
|
||||
)
|
||||
try:
|
||||
with open(absolute_filepath, "wb") as f:
|
||||
f.write(media_data)
|
||||
source_tag["src"] = markdown_path
|
||||
processed_source_successfully = True
|
||||
except IOError as e:
|
||||
source_tag.decompose()
|
||||
else:
|
||||
source_tag.decompose()
|
||||
continue
|
||||
markdown_video_path, _ = self._download_media_file(
|
||||
original_src,
|
||||
base_url_for_relative,
|
||||
self.current_video_absolute_path,
|
||||
markdown_relative_video_folder,
|
||||
media_type_prefixes=("video/", "application/octet-stream"),
|
||||
)
|
||||
if markdown_video_path:
|
||||
source_tag["src"] = markdown_video_path
|
||||
processed_source_successfully = True
|
||||
else:
|
||||
source_tag.decompose()
|
||||
original_video_src_attr = video_tag.get("src")
|
||||
if original_video_src_attr and not processed_source_successfully:
|
||||
original_video_src_attr = original_video_src_attr.strip()
|
||||
if original_video_src_attr:
|
||||
if original_video_src_attr.startswith("cid:"):
|
||||
cid = original_video_src_attr[4:]
|
||||
if (
|
||||
hasattr(self, "mhtml_resources")
|
||||
and cid in self.mhtml_resources
|
||||
):
|
||||
media_data, resource_filename_ext = self.mhtml_resources[
|
||||
cid
|
||||
]
|
||||
ext_from_mhtml = "mp4"
|
||||
if "." in resource_filename_ext:
|
||||
candidate_ext = resource_filename_ext.split(".")[
|
||||
-1
|
||||
].lower()
|
||||
if len(candidate_ext) <= 5 and candidate_ext.isalnum():
|
||||
ext_from_mhtml = candidate_ext
|
||||
unique_filename = f"video_{uuid4().hex}.{ext_from_mhtml}"
|
||||
absolute_filepath = os.path.join(
|
||||
self.current_video_absolute_path, unique_filename
|
||||
)
|
||||
markdown_path = os.path.join(
|
||||
markdown_relative_video_folder, unique_filename
|
||||
)
|
||||
try:
|
||||
with open(absolute_filepath, "wb") as f:
|
||||
f.write(media_data)
|
||||
video_tag["src"] = markdown_path
|
||||
processed_source_successfully = True
|
||||
except IOError as e:
|
||||
if "src" in video_tag.attrs:
|
||||
del video_tag["src"]
|
||||
else:
|
||||
if "src" in video_tag.attrs:
|
||||
del video_tag["src"]
|
||||
else:
|
||||
markdown_video_path, _ = self._download_media_file(
|
||||
original_video_src_attr,
|
||||
base_url_for_relative,
|
||||
self.current_video_absolute_path,
|
||||
markdown_relative_video_folder,
|
||||
media_type_prefixes=("video/", "application/octet-stream"),
|
||||
)
|
||||
if markdown_video_path:
|
||||
video_tag["src"] = markdown_video_path
|
||||
processed_source_successfully = True
|
||||
else:
|
||||
if "src" in video_tag.attrs:
|
||||
del video_tag["src"]
|
||||
|
||||
# If no video source was successfully processed, remove the entire video tag.
|
||||
if not processed_source_successfully:
|
||||
logger.warning(
|
||||
f"Decomposing video tag as no downloadable sources were found."
|
||||
)
|
||||
video_tag.decompose()
|
||||
|
||||
logger.debug("Video processing finished.")
|
||||
return str(soup)
|
||||
|
||||
def _cleanup_markdown(self, markdown_text):
|
||||
logger.debug("Starting Markdown cleanup.")
|
||||
markdown_text = re.sub(r"\[\s*\]\(\s*\)", "", markdown_text)
|
||||
markdown_text = re.sub(r"!\[(.*?)\]\s+\((.*?)\)", r"", markdown_text)
|
||||
markdown_text = re.sub(r"\[(.*?)\]\s+\((.*?)\)", r"[\1](\2)", markdown_text)
|
||||
markdown_text = re.sub(r"\n([*-+])(\S)", r"\n\1 \2", markdown_text)
|
||||
markdown_text = re.sub(r"\n(\d+\.)(\S)", r"\n\1 \2", markdown_text)
|
||||
markdown_text = re.sub(r"!\[\s*\]\((.*?)\)", r"", markdown_text)
|
||||
markdown_text = re.sub(r"\n{3,}", "\n\n", markdown_text)
|
||||
lines = markdown_text.splitlines()
|
||||
stripped_lines = [line.strip() for line in lines]
|
||||
markdown_text = "\n".join(stripped_lines)
|
||||
logger.debug("Markdown cleanup finished.")
|
||||
return markdown_text.strip()
|
||||
|
||||
def convert(self, source, output_filename_stem=None):
|
||||
html_content = None
|
||||
self.base_url = None
|
||||
self.mhtml_resources = {}
|
||||
self.source_html_filepath = None
|
||||
|
||||
if not output_filename_stem:
|
||||
stem = os.path.splitext(os.path.basename(source))[0]
|
||||
output_filename_stem = (
|
||||
stem if stem else f"file_conversion_{uuid4().hex[:8]}"
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Starting conversion. Source: {source}, Type: html_file, Output stem: {output_filename_stem}"
|
||||
)
|
||||
|
||||
self.source_html_filepath = os.path.abspath(
|
||||
source
|
||||
) # Store absolute path of source HTML
|
||||
try:
|
||||
with open(
|
||||
self.source_html_filepath, "r", encoding="utf-8", errors="replace"
|
||||
) as f:
|
||||
html_content = f.read()
|
||||
self.base_url = (
|
||||
Path(self.source_html_filepath).parent.as_uri() + "/"
|
||||
) # file:///path/to/containing_directory/
|
||||
except FileNotFoundError:
|
||||
logger.error(f"HTML file not found: {source}")
|
||||
return None
|
||||
except IOError as e:
|
||||
logger.error(f"Could not read HTML file {source}: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Unexpected error reading HTML file {source}: {e}")
|
||||
return None
|
||||
|
||||
if not html_content:
|
||||
logger.error(f"No HTML content to process from {source}.")
|
||||
return None
|
||||
|
||||
md_img_rel_folder = f"{output_filename_stem}"
|
||||
self.current_image_absolute_path = os.path.join(
|
||||
self.output_dir, md_img_rel_folder
|
||||
)
|
||||
md_vid_rel_folder = f"{output_filename_stem}"
|
||||
self.current_video_absolute_path = os.path.join(
|
||||
self.output_dir, md_vid_rel_folder
|
||||
)
|
||||
try:
|
||||
os.makedirs(self.current_image_absolute_path, exist_ok=True)
|
||||
os.makedirs(self.current_video_absolute_path, exist_ok=True)
|
||||
except OSError as e:
|
||||
logger.error(f"Could not create asset directories: {e}")
|
||||
return None
|
||||
|
||||
logger.info("Cleaning HTML...")
|
||||
cleaned_html = self._clean_html(html_content)
|
||||
|
||||
logger.info("Processing and downloading images...")
|
||||
html_after_images = self._process_images_in_html(
|
||||
cleaned_html, self.base_url, md_img_rel_folder
|
||||
)
|
||||
logger.info("Processing and downloading videos...")
|
||||
# Pass md_img_rel_folder for posters
|
||||
html_after_videos = self._process_videos_in_html(
|
||||
html_after_images, self.base_url, md_vid_rel_folder, md_img_rel_folder
|
||||
)
|
||||
|
||||
logger.info("Converting HTML to Markdown...")
|
||||
try:
|
||||
markdown_output = md(
|
||||
html_after_videos,
|
||||
heading_style="atx",
|
||||
bullets="-",
|
||||
default_title=False,
|
||||
strip=[],
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Error during Markdown conversion for {source}: {e}.")
|
||||
try:
|
||||
markdown_output = md(
|
||||
html_after_images,
|
||||
heading_style="atx",
|
||||
bullets="-",
|
||||
default_title=False,
|
||||
strip=[],
|
||||
) # Fallback
|
||||
except Exception as e2:
|
||||
debug_html_path = os.path.join(
|
||||
self.output_dir, f"{output_filename_stem}_debug_processed.html"
|
||||
)
|
||||
try:
|
||||
with open(debug_html_path, "w", encoding="utf-8") as f_debug:
|
||||
f_debug.write(html_after_videos)
|
||||
except IOError:
|
||||
pass
|
||||
return None
|
||||
|
||||
logger.info("Cleaning Markdown...")
|
||||
final_markdown = self._cleanup_markdown(markdown_output)
|
||||
output_md_path = os.path.join(self.output_dir, f"{output_filename_stem}.md")
|
||||
try:
|
||||
with open(output_md_path, "w", encoding="utf-8") as f:
|
||||
f.write(final_markdown)
|
||||
logger.info(f"Markdown file saved to {output_md_path}")
|
||||
for asset_path in [
|
||||
self.current_image_absolute_path,
|
||||
self.current_video_absolute_path,
|
||||
]:
|
||||
if os.path.exists(asset_path) and not os.listdir(asset_path):
|
||||
try:
|
||||
os.rmdir(asset_path)
|
||||
except OSError as e_rmdir:
|
||||
logger.warning(
|
||||
f"Could not remove empty asset folder {asset_path}: {e_rmdir}"
|
||||
)
|
||||
return output_md_path
|
||||
except IOError as e:
|
||||
logger.error(f"Could not write Markdown file {output_md_path}: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(
|
||||
f"Unexpected error writing Markdown file {output_md_path}: {e}"
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def html_handler(input_file_name, doc_id, converter):
|
||||
if os.path.exists(input_file_name):
|
||||
md_path_local_html = converter.convert(
|
||||
input_file_name,
|
||||
output_filename_stem=doc_id,
|
||||
)
|
||||
if not md_path_local_html:
|
||||
logger.warning(f"Failed to convert local HTML: {input_file_name}")
|
||||
else:
|
||||
logger.debug(f"\nLocal HTML test file not found at '{input_file_name}'.")
|
||||
|
||||
|
||||
def handler(cache_dir, file_or_url: str):
|
||||
output_dir = f"{cache_dir}"
|
||||
|
||||
converter = HTML2MarkdownConverter(
|
||||
output_dir=output_dir,
|
||||
media_download_timeout=60,
|
||||
)
|
||||
|
||||
doc_id = str(uuid4())
|
||||
|
||||
if file_or_url.endswith((".html", ".htm")):
|
||||
html_handler(file_or_url, doc_id, converter)
|
||||
else:
|
||||
logger.error(
|
||||
f"Unsupported file type: {file_or_url}. Only .html and .htm files are supported."
|
||||
)
|
||||
return None, None, None
|
||||
|
||||
return f"{cache_dir}/{doc_id}.md", f"{cache_dir}/{doc_id}", doc_id
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
output_dir = "/Users/tju/Library/Caches/bisheng"
|
||||
input_file = "/Users/tju/Resources/docs/html/f.html"
|
||||
|
||||
output_md, asset_dir, doc_id = handler(output_dir, input_file)
|
||||
if output_md and os.path.exists(output_md):
|
||||
print(f"Markdown saved to: {output_md}")
|
||||
else:
|
||||
print("Conversion failed.")
|
||||
@@ -0,0 +1,182 @@
|
||||
import os
|
||||
import threading
|
||||
from uuid import uuid4
|
||||
|
||||
import fitz
|
||||
from loguru import logger
|
||||
|
||||
pymu_lock = threading.Lock()
|
||||
|
||||
|
||||
def convert_pdf_to_md(output_dir, pdf_path, doc_id):
|
||||
"""
|
||||
将指定的 PDF 文件转换为 Markdown 文件,并保持内容的原有顺序。
|
||||
|
||||
这个函数会提取 PDF 中的文本、表格和图片,并根据它们在页面上的
|
||||
垂直位置进行排序,然后整合到一个 Markdown 文件中。
|
||||
图片会作为独立文件保存在指定的输出目录中。
|
||||
|
||||
Args:
|
||||
pdf_path (str): 输入的 PDF 文件路径。
|
||||
output_dir (str): 保存 Markdown 文件和图片的目录。
|
||||
"""
|
||||
# 确保输出目录存在
|
||||
if not os.path.exists(output_dir):
|
||||
os.makedirs(output_dir)
|
||||
|
||||
md_filename = f"{doc_id}.md"
|
||||
md_filepath = os.path.join(output_dir, md_filename)
|
||||
|
||||
img_dir = os.path.join(output_dir, f"images")
|
||||
if not os.path.exists(img_dir):
|
||||
os.makedirs(img_dir)
|
||||
|
||||
doc = None
|
||||
|
||||
try:
|
||||
doc = fitz.open(pdf_path)
|
||||
except Exception as e:
|
||||
raise Exception('The file is damaged.')
|
||||
try:
|
||||
md_content = ""
|
||||
image_counter = 1
|
||||
|
||||
for page_num in range(len(doc)):
|
||||
with pymu_lock:
|
||||
page = doc.load_page(page_num)
|
||||
|
||||
page_elements = []
|
||||
|
||||
tables = page.find_tables()
|
||||
if tables.tables:
|
||||
for tab in tables.tables:
|
||||
if not tab.to_pandas().empty:
|
||||
md_table = tab.to_pandas().to_markdown(index=False)
|
||||
table_bbox = fitz.Rect(tab.bbox)
|
||||
page_elements.append(
|
||||
{
|
||||
"type": "table",
|
||||
"bbox": table_bbox,
|
||||
"content": md_table,
|
||||
}
|
||||
)
|
||||
|
||||
image_info_list = page.get_image_info(xrefs=True)
|
||||
if image_info_list:
|
||||
for img_info in image_info_list:
|
||||
xref = img_info["xref"]
|
||||
if xref == 0:
|
||||
continue
|
||||
|
||||
base_image = doc.extract_image(xref)
|
||||
if not base_image:
|
||||
continue
|
||||
|
||||
image_bytes = base_image["image"]
|
||||
image_ext = base_image["ext"]
|
||||
|
||||
img_filename = f"image_{page_num + 1}_{image_counter}.{image_ext}"
|
||||
img_path = os.path.join(img_dir, img_filename)
|
||||
|
||||
with open(img_path, "wb") as img_file:
|
||||
img_file.write(image_bytes)
|
||||
|
||||
md_image = f""
|
||||
|
||||
image_bbox = fitz.Rect(img_info["bbox"])
|
||||
page_elements.append(
|
||||
{"type": "image", "bbox": image_bbox, "content": md_image}
|
||||
)
|
||||
image_counter += 1
|
||||
|
||||
table_bboxes = (
|
||||
[fitz.Rect(tab.bbox) for tab in tables.tables]
|
||||
if tables.tables
|
||||
else []
|
||||
)
|
||||
|
||||
text_blocks = page.get_text("blocks")
|
||||
for b in text_blocks:
|
||||
block_rect = fitz.Rect(b[:4])
|
||||
block_text = b[4].strip()
|
||||
|
||||
is_in_table = False
|
||||
for table_bbox in table_bboxes:
|
||||
if block_rect.intersects(table_bbox):
|
||||
is_in_table = True
|
||||
break
|
||||
|
||||
if block_text and not is_in_table:
|
||||
page_elements.append(
|
||||
{"type": "text", "bbox": block_rect, "content": block_text}
|
||||
)
|
||||
|
||||
page_elements.sort(key=lambda el: el["bbox"].y0)
|
||||
|
||||
for elem in page_elements:
|
||||
md_content += elem["content"] + "\n\n"
|
||||
|
||||
with open(md_filepath, "w", encoding="utf-8") as md_file:
|
||||
md_file.write(md_content)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error processing pdf: {e}")
|
||||
raise Exception(f"文档解析失败: {str(e)[-100:]}") # 截取最后100个字符以避免过长的错误信息
|
||||
finally:
|
||||
with pymu_lock:
|
||||
if doc:
|
||||
doc.close()
|
||||
|
||||
|
||||
def is_pdf_damaged(pdf_path: str) -> bool:
|
||||
"""
|
||||
检查 PDF 文件是否损坏。
|
||||
|
||||
Args:
|
||||
pdf_path (str): PDF 文件的路径。
|
||||
|
||||
Returns:
|
||||
bool: 如果文件损坏,返回 True;否则返回 False。
|
||||
"""
|
||||
try:
|
||||
doc = fitz.open(pdf_path)
|
||||
doc.close()
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f"PDF file is damaged: {e}")
|
||||
return True
|
||||
|
||||
|
||||
def handler(cache_dir, file_or_url: str):
|
||||
doc_id = uuid4()
|
||||
ouput_dir = f"{cache_dir}/{doc_id}"
|
||||
convert_pdf_to_md(ouput_dir, file_or_url, doc_id)
|
||||
return f"{ouput_dir}/{doc_id}.md", f"{ouput_dir}/images", doc_id
|
||||
|
||||
|
||||
def exec_thread_safe():
|
||||
pdf_path = "/Users/tju/Documents/Resources/pdf/bisheng/chen4.pdf"
|
||||
output_directory = "/Users/tju/Desktop/output"
|
||||
md_file, local_image, doc_id = handler(output_directory, pdf_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import multiprocessing
|
||||
|
||||
processes = []
|
||||
for _ in range(10):
|
||||
process = multiprocessing.Process(target=exec_thread_safe)
|
||||
processes.append(process)
|
||||
process.start()
|
||||
|
||||
for process in processes:
|
||||
process.join()
|
||||
|
||||
threads = []
|
||||
for i in range(4):
|
||||
thread = threading.Thread(target=exec_thread_safe, name=f"Thread-{i}")
|
||||
threads.append(thread)
|
||||
thread.start()
|
||||
|
||||
for thread in threads:
|
||||
thread.join()
|
||||
@@ -0,0 +1,53 @@
|
||||
from bisheng.pptx2md import convert, ConversionConfig
|
||||
from pathlib import Path
|
||||
from uuid import uuid4
|
||||
|
||||
|
||||
def parser_pptx2md(
|
||||
pptx_file: str,
|
||||
md_file: str,
|
||||
image_dir: str = None,
|
||||
):
|
||||
"""
|
||||
Convert a PowerPoint file to Markdown format.
|
||||
Args:
|
||||
pptx_file (str): Path to the PowerPoint file.
|
||||
md_file (str): Path to the output Markdown file.
|
||||
image_dir (str, optional): Directory to save images. Defaults to None
|
||||
"""
|
||||
# Basic usage
|
||||
convert(
|
||||
ConversionConfig(
|
||||
pptx_path=Path(pptx_file),
|
||||
output_path=Path(md_file),
|
||||
image_dir=Path(image_dir),
|
||||
disable_notes=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def handler(
|
||||
cache_dir,
|
||||
file_name,
|
||||
):
|
||||
doc_id = str(uuid4())
|
||||
md_file_name = f"{cache_dir}/{doc_id}.md"
|
||||
image_dir = f"{cache_dir}/{doc_id}"
|
||||
parser_pptx2md(
|
||||
pptx_file=file_name,
|
||||
md_file=md_file_name,
|
||||
image_dir=image_dir,
|
||||
)
|
||||
# 上传图片,可以用异步
|
||||
# 替换md文件中的图片路径
|
||||
return md_file_name, image_dir, doc_id
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pptx_file = "/Users/tju/Resources/docs/ppt/you-lian.pptx"
|
||||
cache_dir = "/Users/tju/Desktop"
|
||||
|
||||
md_file_name, image_dir, doc_id = handler(cache_dir, pptx_file)
|
||||
print(f"Markdown file: {md_file_name}")
|
||||
print(f"Image directory: {image_dir}")
|
||||
print(f"Document ID: {doc_id}")
|
||||
@@ -0,0 +1,177 @@
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
|
||||
def post_processing(file_path, retain_images=True):
|
||||
"""
|
||||
(最终完整版)
|
||||
全面地将一个Markdown文件中的HTML标签转换为标准Markdown格式,并根据参数正确处理图片。
|
||||
"""
|
||||
try:
|
||||
with open(file_path, "r", encoding="utf-8") as file:
|
||||
content = file.read()
|
||||
|
||||
# 步骤 1: 图片处理 (最优先执行)
|
||||
if not retain_images:
|
||||
# 如果不保留图片,在任何转换前,先全局删除所有格式的图片
|
||||
content = re.sub(r"<img[^>]*>", "", content, flags=re.IGNORECASE)
|
||||
content = re.sub(
|
||||
r"\[!\[.*?\]\(.*?\)\]\(.*?\)", "", content, flags=re.DOTALL
|
||||
)
|
||||
content = re.sub(r"!\[.*?\]\(.*?\)", "", content, flags=re.DOTALL)
|
||||
else:
|
||||
# 如果保留图片,则只转换HTML的img标签为Markdown格式
|
||||
# 使用一个辅助函数来提取src和alt
|
||||
def _img_to_md(match):
|
||||
img_tag = match.group(0)
|
||||
src_match = re.search(r'src="([^"]+)"', img_tag, re.IGNORECASE)
|
||||
alt_match = re.search(r'alt="([^"]*)"', img_tag, re.IGNORECASE)
|
||||
src = src_match.group(1) if src_match else ""
|
||||
alt = alt_match.group(1) if alt_match else ""
|
||||
return f""
|
||||
|
||||
content = re.sub(r"<img[^>]*>", _img_to_md, content, flags=re.IGNORECASE)
|
||||
|
||||
# 步骤 2: 复杂HTML块级元素转换 (使用BeautifulSoup辅助)
|
||||
def _table_to_md(match):
|
||||
soup = BeautifulSoup(match.group(0), "html.parser")
|
||||
headers = [
|
||||
th.get_text(strip=True).replace("|", r"\|")
|
||||
for th in soup.find_all("th")
|
||||
]
|
||||
if not headers: # 如果没有<th>, 尝试把第一行<td>作为表头
|
||||
first_row = soup.find("tr")
|
||||
if not first_row:
|
||||
return ""
|
||||
headers = [
|
||||
td.get_text(strip=True).replace("|", r"\|")
|
||||
for td in first_row.find_all("td")
|
||||
]
|
||||
rows_html = soup.find_all("tr")[1:]
|
||||
else:
|
||||
rows_html = (
|
||||
soup.find("tbody").find_all("tr")
|
||||
if soup.find("tbody")
|
||||
else soup.find_all("tr")[1:]
|
||||
)
|
||||
|
||||
if not headers:
|
||||
return "" # 空表格
|
||||
|
||||
md_table = ["| " + " | ".join(headers) + " |", "|" + "---|" * len(headers)]
|
||||
for row in rows_html:
|
||||
cols = [
|
||||
td.get_text(strip=True).replace("\n", " ").replace("|", r"\|")
|
||||
for td in row.find_all("td")
|
||||
]
|
||||
# 补全单元格以匹配表头长度
|
||||
while len(cols) < len(headers):
|
||||
cols.append("")
|
||||
md_table.append("| " + " | ".join(cols) + " |")
|
||||
return "\n\n" + "\n".join(md_table) + "\n\n"
|
||||
|
||||
content = re.sub(
|
||||
r"<table[^>]*>.*?</table>",
|
||||
_table_to_md,
|
||||
content,
|
||||
flags=re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
|
||||
# 步骤 3: 其他块级和行内HTML标签转换 (主要使用正则)
|
||||
|
||||
# 列表 (简化处理,将ul/ol/li转换为无序列表)
|
||||
content = re.sub(
|
||||
r"<li[^>]*>(.*?)</li>", r"\n- \1", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
content = re.sub(r"</?(ul|ol)[^>]*>", "", content, flags=re.IGNORECASE)
|
||||
# 标题 h1-h6
|
||||
content = re.sub(
|
||||
r"<h([1-6]).*?>(.*?)</h\1>",
|
||||
lambda m: "\n" + "#" * int(m.group(1)) + " " + m.group(2).strip() + "\n",
|
||||
content,
|
||||
flags=re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
# 引用
|
||||
content = re.sub(
|
||||
r"<blockquote[^>]*>(.*?)</blockquote>",
|
||||
lambda m: "\n> " + m.group(1).strip().replace("\n", "\n> ") + "\n",
|
||||
content,
|
||||
flags=re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
# 链接
|
||||
content = re.sub(
|
||||
r'<a\s+href="([^"]+)"[^>]*>(.*?)</a>',
|
||||
r"[\2](\1)",
|
||||
content,
|
||||
flags=re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
# 加粗
|
||||
content = re.sub(
|
||||
r"<(strong|b)>(.*?)</\1>",
|
||||
r"**\2**",
|
||||
content,
|
||||
flags=re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
# 斜体
|
||||
content = re.sub(
|
||||
r"<(em|i)>(.*?)</\1>", r"*\2*", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
# 删除线
|
||||
content = re.sub(
|
||||
r"<(del|s)>(.*?)</\1>", r"~~\2~~", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
# 上标/下标
|
||||
content = re.sub(
|
||||
r"<sup>(.*?)</sup>", r"^\1^", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
content = re.sub(
|
||||
r"<sub>(.*?)</sub>", r"~\1~", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
# 行内代码
|
||||
content = re.sub(
|
||||
r"<code>(.*?)</code>", r"`\1`", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
# 水平线
|
||||
content = re.sub(r"<hr[^>]*>", "\n---\n", content, flags=re.IGNORECASE)
|
||||
# 换行
|
||||
content = re.sub(r"<br\s*/?>", " \n", content, flags=re.IGNORECASE)
|
||||
# 段落 (转换为换行)
|
||||
content = re.sub(r"</p>", "\n", content, flags=re.IGNORECASE)
|
||||
content = re.sub(r"<p[^>]*>", "\n", content, flags=re.IGNORECASE)
|
||||
# Span (移除标签,保留内容)
|
||||
content = re.sub(
|
||||
r"<span[^>]*>(.*?)</span>", r"\1", content, flags=re.IGNORECASE | re.DOTALL
|
||||
)
|
||||
|
||||
# 步骤 4: 最终清理
|
||||
content = html.unescape(content) # 解码HTML实体
|
||||
content = re.sub(r"\n{3,}", "\n\n", content.strip()) # 规范化空行
|
||||
|
||||
with open(file_path, "w", encoding="utf-8") as file:
|
||||
file.write(content)
|
||||
|
||||
except FileNotFoundError:
|
||||
raise Exception(f"错误: 文件 {file_path} 未找到。")
|
||||
except Exception as e:
|
||||
raise Exception(f"处理文件时发生错误: {e}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# --- 使用方法 ---
|
||||
# 请将下面的路径替换为您要处理的.md文件的实际路径
|
||||
# markdown_file_to_process = "/path/to/your/markdown_file.md"
|
||||
markdown_file_to_process = (
|
||||
"/Users/tju/Desktop/d40c526e-2081-49c3-9603-83132ce88978.md" # 示例,请替换
|
||||
)
|
||||
|
||||
if os.path.exists(markdown_file_to_process):
|
||||
# 示例1: 转换HTML并保留图片
|
||||
post_processing(markdown_file_to_process, retain_images=True)
|
||||
|
||||
# 示例2: 转换HTML并移除所有图片
|
||||
# post_processing_revised(markdown_file_to_process, retain_images=False)
|
||||
else:
|
||||
print(f"请将脚本中的 'your_markdown_file.md' 替换为真实的文件路径后再运行。")
|
||||
@@ -1,17 +1,19 @@
|
||||
import json
|
||||
|
||||
from bisheng.database.models.gpts_tools import AuthMethod
|
||||
from bisheng.database.models.gpts_tools import AuthMethod, AuthType
|
||||
|
||||
|
||||
class OpenApiSchema:
|
||||
|
||||
def __init__(self, contents: dict):
|
||||
self.contents = contents
|
||||
self.version = contents['openapi']
|
||||
self.info = contents['info']
|
||||
self.title = self.info['title']
|
||||
self.auth_type = 'basic'
|
||||
self.auth_method = 0
|
||||
self.description = self.info.get('description', '')
|
||||
|
||||
self.default_server = ""
|
||||
self.default_server = ''
|
||||
self.apis = []
|
||||
|
||||
def parse_server(self) -> str:
|
||||
@@ -25,41 +27,112 @@ class OpenApiSchema:
|
||||
self.default_server = servers[0]['url']
|
||||
else:
|
||||
self.default_server = servers['url']
|
||||
|
||||
# if self.contents.get('components') and self.contents['components'].get('securitySchemes') is not None:
|
||||
# self.auth_type = 'custom' if self.contents['components']['securitySchemes']['ApiKeyAuth']['type'] == 'apiKey' else 'basic'
|
||||
# s = self.contents['components']['securitySchemes']['ApiKeyAuth']['schema']
|
||||
# if self.contents['components']['securitySchemes']['ApiKeyAuth']['type'] == 'http':
|
||||
# self.auth_type = s
|
||||
#
|
||||
# self.auth_method= 1 if self.contents['components']['securitySchemes']['ApiKeyAuth']['type'] == 'apiKey' or 'http' else 0
|
||||
# self.api_location= self.contents['components']['securitySchemes']['ApiKeyAuth']['in']
|
||||
# self.parameter_name= self.contents['components']['securitySchemes']['ApiKeyAuth']['name']
|
||||
|
||||
security_schemes = self.contents.get('components', {}).get('securitySchemes', {})
|
||||
api_key_auth = security_schemes.get('ApiKeyAuth', {})
|
||||
|
||||
# 获取认证类型
|
||||
auth_type = api_key_auth.get('type')
|
||||
if auth_type == 'apiKey':
|
||||
self.auth_type = 'custom'
|
||||
elif auth_type == 'http':
|
||||
self.auth_type = api_key_auth.get('schema')
|
||||
else:
|
||||
self.auth_type = 'basic'
|
||||
|
||||
# 设置认证方法
|
||||
self.auth_method = 1 if auth_type in ('apiKey', 'http') else 0
|
||||
|
||||
# 获取 API 位置和参数名
|
||||
self.api_location = api_key_auth.get('in')
|
||||
self.parameter_name = api_key_auth.get('name')
|
||||
return self.default_server
|
||||
|
||||
def parse_paths(self) -> list[dict]:
|
||||
paths = self.contents['paths']
|
||||
|
||||
self.apis = []
|
||||
|
||||
for path, path_info in paths.items():
|
||||
for method, method_info in path_info.items():
|
||||
one_api_info = {
|
||||
"path": path,
|
||||
"method": method
|
||||
'path': path,
|
||||
'method': method,
|
||||
'description': method_info.get('description', '')
|
||||
or method_info.get('summary', ''),
|
||||
'operationId': method_info['operationId'],
|
||||
'parameters': [],
|
||||
}
|
||||
if method not in ['get', 'post', 'put', 'delete']:
|
||||
continue
|
||||
one_api_info["description"] = method_info.get('description', '') or method_info.get('summary', '')
|
||||
one_api_info["operationId"] = method_info['operationId']
|
||||
one_api_info["parameters"] = method_info.get('parameters', [])
|
||||
|
||||
if 'requestBody' in method_info:
|
||||
for _, content in method_info['requestBody']['content'].items():
|
||||
if '$ref' in content['schema']:
|
||||
schema_ref = content['schema']['$ref']
|
||||
schema_name = schema_ref.split('/')[-1]
|
||||
schema = self.contents['components']['schemas'][schema_name]
|
||||
else:
|
||||
schema = content['schema']
|
||||
|
||||
if 'properties' in schema:
|
||||
for param_name, param_info in schema['properties'].items():
|
||||
param = {
|
||||
'name': param_name,
|
||||
'description': param_info.get('description', ''),
|
||||
'in': 'body',
|
||||
'required': param_name in schema.get('required', []),
|
||||
'schema': {
|
||||
'type': param_info.get('type', 'string'),
|
||||
'title': param_info.get('title', param_name),
|
||||
'properties': param_info.get('properties', {})
|
||||
},
|
||||
}
|
||||
one_api_info['parameters'].append(param)
|
||||
else:
|
||||
# no request body get parameters
|
||||
one_api_info['parameters'].extend(method_info.get('parameters', []))
|
||||
self.apis.append(one_api_info)
|
||||
return self.apis
|
||||
|
||||
@staticmethod
|
||||
def parse_openapi_tool_params(name: str, description: str, extra: str, server_host: str,
|
||||
auth_method: int, auth_type: str = None, api_key: str = None):
|
||||
def parse_openapi_tool_params(name: str,
|
||||
description: str,
|
||||
extra: str,
|
||||
server_host: str,
|
||||
auth_method: int,
|
||||
auth_type: str = None,
|
||||
api_key: str = None):
|
||||
# 拼接请求头
|
||||
headers = {}
|
||||
if auth_method == AuthMethod.API_KEY.value:
|
||||
headers = {
|
||||
"Authorization": f"{auth_type} {api_key}"
|
||||
}
|
||||
if auth_type == AuthType.CUSTOM.value:
|
||||
extra_json = json.loads(extra)
|
||||
location = extra_json["api_location"]
|
||||
parameter_name= extra_json["parameter_name"]
|
||||
if location == "header":
|
||||
headers = {parameter_name: api_key}
|
||||
elif auth_type == AuthType.BASIC.value:
|
||||
headers = {'Authorization': f'Basic {api_key}'}
|
||||
elif auth_type == AuthType.BEARER.value:
|
||||
headers = {'Authorization': f'Bearer {api_key}'}
|
||||
|
||||
# 返回初始化 openapi所需的入参
|
||||
params = {
|
||||
"params": json.loads(extra),
|
||||
"headers": headers,
|
||||
"url": server_host,
|
||||
"description": name + description if description else name
|
||||
'params': json.loads(extra),
|
||||
'headers': headers,
|
||||
'api_key': api_key,
|
||||
'url': server_host,
|
||||
'description': name + description if description else name
|
||||
}
|
||||
return params
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
import os
|
||||
|
||||
from langchain_core.documents import Document
|
||||
|
||||
from bisheng.api.services.md_from_docx import handler as docx_handler
|
||||
from bisheng.api.services.md_from_excel import handler as excel_handler
|
||||
from bisheng.api.services.md_from_html import handler as html_handler
|
||||
from bisheng.api.services.md_from_pdf import handler as pdf_handler
|
||||
from bisheng.api.services.md_from_pptx import handler as pptx_handler
|
||||
from bisheng.api.services.md_post_processing import post_processing
|
||||
from bisheng.cache.utils import CACHE_DIR
|
||||
from bisheng.utils.minio_client import minio_client
|
||||
|
||||
|
||||
def combine_multiple_md_files_to_raw_texts(
|
||||
path,
|
||||
) -> tuple[list[Document], list[Document]]:
|
||||
"""
|
||||
combine multiple md file to raw texts including meta-data list.
|
||||
Args:
|
||||
path: the directory containing the md files.
|
||||
Returns:
|
||||
0: split raw texts, each text is a Document object.
|
||||
1: a single Document object containing all the texts combined.
|
||||
"""
|
||||
|
||||
files = sorted([f for f in os.listdir(path)])
|
||||
raw_texts = []
|
||||
|
||||
# 一个文件只对应一个完整的 Document 对象, texts 才是切分后的chunk内容
|
||||
documents = [Document(page_content="", metadata={})]
|
||||
|
||||
for file_name in files:
|
||||
full_file_name = f"{path}/{file_name}"
|
||||
with open(full_file_name, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
raw_texts.append(Document(page_content=content, metadata={}))
|
||||
documents[0].page_content += content
|
||||
return raw_texts, documents
|
||||
|
||||
|
||||
def convert_file_to_md(
|
||||
file_name,
|
||||
input_file_name,
|
||||
header_rows=[0, 1],
|
||||
data_rows=10,
|
||||
append_header=True,
|
||||
knowledge_id=None,
|
||||
retain_images=True,
|
||||
):
|
||||
"""
|
||||
处理文件转换的主函数。
|
||||
Args:
|
||||
file_name:
|
||||
input_file_name:
|
||||
header_rows:
|
||||
data_rows:
|
||||
append_header:
|
||||
knowledge_id:
|
||||
"""
|
||||
md_file_name = None
|
||||
local_image_dir = None
|
||||
include_cache_dir = True
|
||||
doc_id = None
|
||||
if file_name.endswith(".docx") or file_name.endswith(".doc"):
|
||||
md_file_name, local_image_dir, doc_id = docx_handler(CACHE_DIR, input_file_name)
|
||||
elif file_name.endswith(".pptx") or file_name.endswith(".ppt"):
|
||||
md_file_name, local_image_dir, doc_id = pptx_handler(CACHE_DIR, input_file_name)
|
||||
include_cache_dir = False
|
||||
elif (
|
||||
file_name.endswith(".xlsx")
|
||||
or file_name.endswith(".xls")
|
||||
or file_name.endswith(".csv")
|
||||
):
|
||||
md_file_name, local_image_dir, doc_id = excel_handler(
|
||||
CACHE_DIR, input_file_name, header_rows, data_rows, append_header
|
||||
)
|
||||
local_image_dir = None
|
||||
return md_file_name, local_image_dir, doc_id
|
||||
elif (
|
||||
file_name.endswith(".html")
|
||||
or file_name.endswith(".htm")
|
||||
or file_name.endswith(".mhtml")
|
||||
):
|
||||
(
|
||||
md_file_name,
|
||||
local_image_dir,
|
||||
doc_id,
|
||||
) = html_handler(CACHE_DIR, input_file_name)
|
||||
include_cache_dir = False
|
||||
elif file_name.endswith("pdf"):
|
||||
md_file_name, local_image_dir, doc_id = pdf_handler(CACHE_DIR, input_file_name)
|
||||
include_cache_dir = True
|
||||
|
||||
return replace_image_url(
|
||||
md_file_name,
|
||||
local_image_dir,
|
||||
doc_id,
|
||||
include_cache_dir,
|
||||
knowledge_id=knowledge_id,
|
||||
retain_images=retain_images,
|
||||
)
|
||||
|
||||
|
||||
def replace_image_url(
|
||||
md_file_name,
|
||||
local_image_dir,
|
||||
doc_id,
|
||||
include_cache_dir,
|
||||
knowledge_id=None,
|
||||
retain_images=True,
|
||||
):
|
||||
"""
|
||||
Usage:
|
||||
user the same bucket as origin file located.
|
||||
Args:
|
||||
md_file_name:
|
||||
local_image_dir:
|
||||
doc_id:
|
||||
knowledge_id:
|
||||
if the knowledge_id is None, this process will be interrupted,
|
||||
because the image files wouldn't be put into minio
|
||||
"""
|
||||
from bisheng.api.services.knowledge_imp import KnowledgeUtils
|
||||
|
||||
minio_image_path = f"/{minio_client.bucket}/{KnowledgeUtils.get_knowledge_file_image_dir(doc_id, knowledge_id)}"
|
||||
url_for_replacement = local_image_dir
|
||||
if not include_cache_dir:
|
||||
url_for_replacement = doc_id
|
||||
|
||||
if md_file_name and local_image_dir and doc_id:
|
||||
with open(md_file_name, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
content = content.replace(url_for_replacement, minio_image_path)
|
||||
|
||||
with open(md_file_name, "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
post_processing(md_file_name, retain_images)
|
||||
return md_file_name, local_image_dir, doc_id
|
||||
@@ -0,0 +1,356 @@
|
||||
import json
|
||||
from datetime import datetime
|
||||
from typing import List, Any, Dict, Optional
|
||||
|
||||
from fastapi.encoders import jsonable_encoder
|
||||
from fastapi import Request, HTTPException
|
||||
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.api.services.assistant import AssistantService
|
||||
from bisheng.api.services.audit_log import AuditLogService
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.errcode.user import UserGroupNotDeleteError
|
||||
from bisheng.api.utils import get_request_ip
|
||||
from bisheng.api.v1.schemas import resp_200
|
||||
from bisheng.database.constants import AdminRole
|
||||
from bisheng.database.models.assistant import AssistantDao
|
||||
from bisheng.database.models.flow import FlowDao, FlowType
|
||||
from bisheng.database.models.gpts_tools import GptsToolsDao
|
||||
from bisheng.database.models.group import Group, GroupCreate, GroupDao, GroupRead, DefaultGroup
|
||||
from bisheng.database.models.group_resource import GroupResourceDao, ResourceTypeEnum
|
||||
from bisheng.database.models.knowledge import KnowledgeDao
|
||||
from bisheng.database.models.role import RoleDao
|
||||
from bisheng.database.models.user import User, UserDao
|
||||
from bisheng.database.models.user_role import UserRoleDao
|
||||
from bisheng.database.models.user_group import UserGroupCreate, UserGroupDao, UserGroupRead
|
||||
from loguru import logger
|
||||
|
||||
|
||||
class RoleGroupService():
|
||||
|
||||
def get_group_list(self, group_ids: List[int]) -> List[GroupRead]:
|
||||
"""获取全量的group列表"""
|
||||
|
||||
# 查询group
|
||||
if group_ids:
|
||||
groups = GroupDao.get_group_by_ids(group_ids)
|
||||
else:
|
||||
groups = GroupDao.get_all_group()
|
||||
# 查询user
|
||||
user_admin = UserGroupDao.get_groups_admins([group.id for group in groups])
|
||||
users_dict = {}
|
||||
if user_admin:
|
||||
user_ids = [user.user_id for user in user_admin]
|
||||
users = UserDao.get_user_by_ids(user_ids)
|
||||
users_dict = {user.user_id: user for user in users}
|
||||
|
||||
groupReads = [GroupRead.validate(group) for group in groups]
|
||||
for group in groupReads:
|
||||
group.group_admins = [
|
||||
users_dict.get(user.user_id).model_dump() for user in user_admin
|
||||
if user.group_id == group.id
|
||||
]
|
||||
return groupReads
|
||||
|
||||
def create_group(self, request: Request, login_user: UserPayload, group: GroupCreate) -> Group:
|
||||
"""新建用户组"""
|
||||
group_admin = group.group_admins
|
||||
group.create_user = login_user.user_id
|
||||
group.update_user = login_user.user_id
|
||||
group = GroupDao.insert_group(group)
|
||||
if group_admin:
|
||||
logger.info('set_admin group_admins={} group_id={}', group_admin, group.id)
|
||||
self.set_group_admin(request, login_user, group_admin, group.id)
|
||||
self.create_group_hook(request, login_user, group)
|
||||
return group
|
||||
|
||||
def create_group_hook(self, request: Request, login_user: UserPayload, group: Group) -> bool:
|
||||
""" 新建用户组后置操作 """
|
||||
logger.info(f'act=create_group_hook user={login_user.user_name} group_id={group.id}')
|
||||
# 记录审计日志
|
||||
AuditLogService.create_user_group(login_user, get_request_ip(request), group)
|
||||
return True
|
||||
|
||||
def update_group(self, request: Request, login_user: UserPayload, group: Group) -> Group:
|
||||
"""更新用户组"""
|
||||
exist_group = GroupDao.get_user_group(group.id)
|
||||
if not exist_group:
|
||||
raise ValueError('用户组不存在')
|
||||
exist_group.group_name = group.group_name
|
||||
exist_group.remark = group.group_name
|
||||
exist_group.update_user = login_user.user_id
|
||||
exist_group.update_time = datetime.now()
|
||||
|
||||
group = GroupDao.update_group(exist_group)
|
||||
self.update_group_hook(request, login_user, group)
|
||||
return group
|
||||
|
||||
def update_group_hook(self, request: Request, login_user: UserPayload, group: Group):
|
||||
logger.info(f'act=update_group_hook user={login_user.user_name} group_id={group.id}')
|
||||
# 记录审计日志
|
||||
AuditLogService.update_user_group(login_user, get_request_ip(request), group)
|
||||
|
||||
def delete_group(self, request: Request, login_user: UserPayload, group_id: int):
|
||||
"""删除用户组"""
|
||||
if group_id == DefaultGroup:
|
||||
raise HTTPException(status_code=500, detail='默认组不能删除')
|
||||
group_info = GroupDao.get_user_group(group_id)
|
||||
if not group_info:
|
||||
return resp_200()
|
||||
|
||||
# 判断组下是否还有用户
|
||||
user_group_list = UserGroupDao.get_group_user(group_id)
|
||||
if user_group_list:
|
||||
return UserGroupNotDeleteError.return_resp()
|
||||
GroupDao.delete_group(group_id)
|
||||
self.delete_group_hook(request, login_user, group_info)
|
||||
return resp_200()
|
||||
|
||||
def delete_group_hook(self, request: Request, login_user: UserPayload, group_info: Group):
|
||||
logger.info(f'act=delete_group_hook user={login_user.user_name} group_id={group_info.id}')
|
||||
# 记录审计日志
|
||||
AuditLogService.delete_user_group(login_user, get_request_ip(request), group_info)
|
||||
# 将组下资源移到默认用户组
|
||||
# 获取组下所有的资源
|
||||
all_resource = GroupResourceDao.get_group_all_resource(group_info.id)
|
||||
need_move_resource = []
|
||||
for one in all_resource:
|
||||
# 获取资源属于几个组,属于多个组则不用处理, 否则将资源转移到默认用户组
|
||||
resource_groups = GroupResourceDao.get_resource_group(ResourceTypeEnum(one.type), one.third_id)
|
||||
if len(resource_groups) > 1:
|
||||
continue
|
||||
else:
|
||||
one.group_id = DefaultGroup
|
||||
need_move_resource.append(one)
|
||||
if need_move_resource:
|
||||
GroupResourceDao.update_group_resource(need_move_resource)
|
||||
GroupResourceDao.delete_group_resource_by_group_id(group_info.id)
|
||||
# 删除用户组下的角色列表
|
||||
RoleDao.delete_role_by_group_id(group_info.id)
|
||||
# 删除用户组的管理员
|
||||
UserGroupDao.delete_group_all_admin(group_info.id)
|
||||
# 将删除事件发到redis队列中
|
||||
delete_message = json.dumps({"id": group_info.id})
|
||||
redis_client.rpush('delete_group', delete_message, expiration=86400)
|
||||
redis_client.publish('delete_group', delete_message)
|
||||
|
||||
def get_group_user_list(self, group_id: int, page_size: int, page_num: int) -> List[User]:
|
||||
"""获取全量的group列表"""
|
||||
|
||||
# 查询user
|
||||
user_group_list = UserGroupDao.get_group_user(group_id, page_size, page_num)
|
||||
if user_group_list:
|
||||
user_ids = [user.user_id for user in user_group_list]
|
||||
return UserDao.get_user_by_ids(user_ids)
|
||||
|
||||
return None
|
||||
|
||||
def insert_user_group(self, user_group: UserGroupCreate) -> UserGroupRead:
|
||||
"""插入用户组"""
|
||||
|
||||
user_groups = UserGroupDao.get_user_group(user_group.user_id)
|
||||
if user_groups and user_group.group_id in [ug.group_id for ug in user_groups]:
|
||||
raise ValueError('重复设置用户组')
|
||||
|
||||
return UserGroupDao.insert_user_group(user_group)
|
||||
|
||||
def replace_user_groups(self, request: Request, login_user: UserPayload, user_id: int, group_ids: List[int]):
|
||||
""" 覆盖用户的所在的用户组 """
|
||||
# 判断下被操作用户是否是超级管理员
|
||||
user_role_list = UserRoleDao.get_user_roles(user_id)
|
||||
if any(one.role_id == AdminRole for one in user_role_list):
|
||||
raise HTTPException(status_code=500, detail='系统管理员不允许编辑')
|
||||
|
||||
# 获取用户之前的所有分组
|
||||
old_group = UserGroupDao.get_user_group(user_id)
|
||||
old_group = [one.group_id for one in old_group]
|
||||
if not login_user.is_admin():
|
||||
# 获取操作人所管理的组
|
||||
admin_group = UserGroupDao.get_user_admin_group(login_user.user_id)
|
||||
admin_group = [one.group_id for one in admin_group]
|
||||
# 过滤被操作人所在的组,只处理有权限管理的组
|
||||
old_group = [one for one in old_group if one in admin_group]
|
||||
# 说明此用户 不在此用户组管理员所管辖的用户组内
|
||||
if not old_group:
|
||||
raise ValueError('没有权限设置用户组')
|
||||
need_delete_group = old_group.copy()
|
||||
need_add_group = []
|
||||
for one in group_ids:
|
||||
if one not in old_group:
|
||||
# 需要加入的用户组
|
||||
need_add_group.append(one)
|
||||
else:
|
||||
# 旧的用户组里剩余的就是要移出的用户组
|
||||
need_delete_group.remove(one)
|
||||
if need_delete_group:
|
||||
UserGroupDao.delete_user_groups(user_id, need_delete_group)
|
||||
if need_add_group:
|
||||
UserGroupDao.add_user_groups(user_id, need_add_group)
|
||||
|
||||
# 记录审计日志
|
||||
group_infos = GroupDao.get_group_by_ids(old_group + group_ids)
|
||||
group_dict: Dict[int, str] = {}
|
||||
for one in group_infos:
|
||||
group_dict[one.id] = one.group_name
|
||||
note = "编辑前用户组:"
|
||||
for one in old_group:
|
||||
note += f'{group_dict.get(one, one)}、'
|
||||
note = note.rstrip('、')
|
||||
note += "编辑后用户组:"
|
||||
for one in group_ids:
|
||||
note += f'{group_dict.get(one, one)}、'
|
||||
note = note.rstrip('、')
|
||||
AuditLogService.update_user(login_user, get_request_ip(request), user_id, list(group_dict.keys()), note)
|
||||
return None
|
||||
|
||||
def get_user_groups_list(self, user_id: int) -> List[GroupRead]:
|
||||
"""获取用户组列表"""
|
||||
user_groups = UserGroupDao.get_user_group(user_id)
|
||||
if not user_groups:
|
||||
return []
|
||||
group_ids = [ug.group_id for ug in user_groups]
|
||||
return GroupDao.get_group_by_ids(group_ids)
|
||||
|
||||
def set_group_admin(self, request: Request, login_user: UserPayload, user_ids: List[int], group_id: int):
|
||||
"""设置用户组管理员"""
|
||||
# 获取目前用户组的管理员列表
|
||||
user_group_admins = UserGroupDao.get_groups_admins([group_id])
|
||||
res = []
|
||||
need_delete_admin = []
|
||||
need_add_admin = user_ids
|
||||
if user_group_admins:
|
||||
for user in user_group_admins:
|
||||
if user.user_id in need_add_admin:
|
||||
res.append(user)
|
||||
need_add_admin.remove(user.user_id)
|
||||
else:
|
||||
need_delete_admin.append(user.user_id)
|
||||
if need_add_admin:
|
||||
# 可以分配非组内用户为管理员。进行用户创建
|
||||
for user_id in need_add_admin:
|
||||
res.append(UserGroupDao.insert_user_group_admin(user_id, group_id))
|
||||
if need_delete_admin:
|
||||
UserGroupDao.delete_group_admins(group_id, need_delete_admin)
|
||||
# 修改用户组的最近修改人
|
||||
GroupDao.update_group_update_user(group_id, login_user.user_id)
|
||||
|
||||
group_info = GroupDao.get_user_group(group_id)
|
||||
self.update_group_hook(request, login_user, group_info)
|
||||
return res
|
||||
|
||||
def set_group_update_user(self, login_user: UserPayload, group_id: int):
|
||||
"""设置用户组管理员"""
|
||||
GroupDao.update_group_update_user(group_id, login_user.user_id)
|
||||
|
||||
def get_group_resources(self, group_id: int, resource_type: ResourceTypeEnum, name: str,
|
||||
page_size: int, page_num: int) -> (List[Any], int):
|
||||
""" 获取用户下的资源 """
|
||||
if resource_type.value == ResourceTypeEnum.FLOW.value:
|
||||
return self.get_group_flow(group_id, name, page_size, page_num)
|
||||
elif resource_type.value == ResourceTypeEnum.KNOWLEDGE.value:
|
||||
return self.get_group_knowledge(group_id, name, page_size, page_num)
|
||||
elif resource_type.value == ResourceTypeEnum.WORK_FLOW.value:
|
||||
return self.get_group_flow(group_id, name, page_size, page_num, FlowType.WORKFLOW)
|
||||
elif resource_type.value == ResourceTypeEnum.ASSISTANT.value:
|
||||
return self.get_group_assistant(group_id, name, page_size, page_num)
|
||||
elif resource_type.value == ResourceTypeEnum.GPTS_TOOL.value:
|
||||
return self.get_group_tool(group_id, name, page_size, page_num)
|
||||
logger.warning('not support resource type: %s', resource_type)
|
||||
return [], 0
|
||||
|
||||
def get_user_map(self, user_ids: set[int]):
|
||||
user_list = UserDao.get_user_by_ids(list(user_ids))
|
||||
user_map = {user.user_id: user.user_name for user in user_list}
|
||||
return user_map
|
||||
|
||||
def get_group_flow(self, group_id: int, keyword: str, page_size: int, page_num: int,flow_type:Optional[FlowType] = None) -> (List[Any], int):
|
||||
""" 获取用户组下的知识库列表 """
|
||||
# 查询用户组下的技能ID列表
|
||||
rs_type = ResourceTypeEnum.FLOW
|
||||
if flow_type == FlowType.WORKFLOW:
|
||||
rs_type = ResourceTypeEnum.WORK_FLOW
|
||||
resource_list = GroupResourceDao.get_group_resource(group_id, rs_type)
|
||||
if not resource_list:
|
||||
return [], 0
|
||||
res = []
|
||||
flow_ids = [resource.third_id for resource in resource_list]
|
||||
flow_type_value = flow_type.value if flow_type else FlowType.FLOW.value
|
||||
data, total = FlowDao.filter_flows_by_ids(flow_ids, keyword, page_num, page_size, flow_type_value)
|
||||
db_user_ids = {one.user_id for one in data}
|
||||
user_map = self.get_user_map(db_user_ids)
|
||||
for one in data:
|
||||
one_dict = jsonable_encoder(one)
|
||||
one_dict["user_name"] = user_map.get(one.user_id, one.user_id)
|
||||
res.append(one_dict)
|
||||
|
||||
return res, total
|
||||
|
||||
def get_group_knowledge(self, group_id: int, keyword: str, page_size: int, page_num: int) -> (List[Any], int):
|
||||
""" 获取用户组下的知识库列表 """
|
||||
# 查询用户组下的知识库ID列表
|
||||
resource_list = GroupResourceDao.get_group_resource(group_id, ResourceTypeEnum.KNOWLEDGE)
|
||||
if not resource_list:
|
||||
return [], 0
|
||||
res = []
|
||||
knowledge_ids = [int(resource.third_id) for resource in resource_list]
|
||||
# 查询知识库
|
||||
data, total = KnowledgeDao.filter_knowledge_by_ids(knowledge_ids, keyword, page_num, page_size)
|
||||
db_user_ids = {one.user_id for one in data}
|
||||
user_map = self.get_user_map(db_user_ids)
|
||||
for one in data:
|
||||
one_dict = jsonable_encoder(one)
|
||||
one_dict["user_name"] = user_map.get(one.user_id, one.user_id)
|
||||
res.append(one_dict)
|
||||
return res, total
|
||||
|
||||
def get_group_assistant(self, group_id: int, keyword: str, page_size: int, page_num: int) -> (List[Any], int):
|
||||
""" 获取用户组下的助手列表 """
|
||||
# 查询用户组下的助手ID列表
|
||||
resource_list = GroupResourceDao.get_group_resource(group_id, ResourceTypeEnum.ASSISTANT)
|
||||
if not resource_list:
|
||||
return [], 0
|
||||
res = []
|
||||
assistant_ids = [resource.third_id for resource in resource_list] # 查询助手
|
||||
data, total = AssistantDao.filter_assistant_by_id(assistant_ids, keyword, page_num, page_size)
|
||||
for one in data:
|
||||
simple_one = AssistantService.return_simple_assistant_info(one)
|
||||
res.append(simple_one)
|
||||
return res, total
|
||||
|
||||
def get_group_tool(self, group_id: int, keyword: str, page_size: int, page_num: int) -> (List[Any], int):
|
||||
""" 获取用户组下的工具列表 """
|
||||
# 查询用户组下的工具ID列表
|
||||
resource_list = GroupResourceDao.get_group_resource(group_id, ResourceTypeEnum.GPTS_TOOL)
|
||||
if not resource_list:
|
||||
return [], 0
|
||||
res = []
|
||||
tool_ids = [int(resource.third_id) for resource in resource_list]
|
||||
# 查询工具
|
||||
data, total = GptsToolsDao.filter_tool_types_by_ids(tool_ids, keyword, page_num, page_size)
|
||||
db_user_ids = {one.user_id for one in data}
|
||||
user_map = self.get_user_map(db_user_ids)
|
||||
for one in data:
|
||||
one_dict = jsonable_encoder(one)
|
||||
one_dict["user_name"] = user_map.get(one.user_id, one.user_id)
|
||||
res.append(one_dict)
|
||||
return res, total
|
||||
|
||||
def get_manage_resources(self, login_user: UserPayload, keyword: str, page: int, page_size: int) -> (list, int):
|
||||
""" 获取用户所管理的用户组下的应用列表 包含技能、助手、工作流"""
|
||||
groups = []
|
||||
if not login_user.is_admin():
|
||||
groups = [str(one.group_id) for one in UserGroupDao.get_user_admin_group(login_user.user_id)]
|
||||
if not groups:
|
||||
return [], 0
|
||||
|
||||
resource_ids = []
|
||||
# 说明是用户组管理员,需要过滤获取到对应组下的资源
|
||||
if groups:
|
||||
group_resources = GroupResourceDao.get_groups_resource(groups, resource_types=[ResourceTypeEnum.FLOW,
|
||||
ResourceTypeEnum.ASSISTANT,
|
||||
ResourceTypeEnum.WORK_FLOW])
|
||||
if not group_resources:
|
||||
return [], 0
|
||||
resource_ids = [one.third_id for one in group_resources]
|
||||
|
||||
return FlowDao.get_all_apps(keyword, id_list=resource_ids, page=page, limit=page_size)
|
||||
@@ -1,4 +1,4 @@
|
||||
from typing import Dict
|
||||
from typing import Dict, List
|
||||
|
||||
import requests
|
||||
|
||||
@@ -119,6 +119,13 @@ class SFTBackend:
|
||||
'model_name': model_name})
|
||||
return cls.handle_response(res)
|
||||
|
||||
@classmethod
|
||||
def get_all_model(cls, host) -> (bool, List[str]):
|
||||
""" 获取所有的模型列表 """
|
||||
url = '/v2.1/sft/model'
|
||||
res = requests.get(f'{host}{url}')
|
||||
return cls.handle_response(res)
|
||||
|
||||
@classmethod
|
||||
def get_gpu_info(cls, host) -> (bool, str):
|
||||
""" 获取GPU信息 """
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
import json
|
||||
from typing import List
|
||||
|
||||
from fastapi import Request, HTTPException
|
||||
from loguru import logger
|
||||
|
||||
from bisheng.api.errcode.base import UnAuthorizedError
|
||||
from bisheng.api.errcode.tag import TagExistError, TagNotExistError
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.database.models.assistant import AssistantDao
|
||||
from bisheng.database.models.config import ConfigDao, ConfigKeyEnum, Config
|
||||
from bisheng.database.models.flow import FlowDao
|
||||
from bisheng.database.models.group_resource import ResourceTypeEnum, GroupResourceDao
|
||||
from bisheng.database.models.tag import TagDao, Tag, TagLink
|
||||
|
||||
|
||||
class TagService:
|
||||
|
||||
@classmethod
|
||||
def get_all_tag(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
keyword: str = None, page: int = 0, limit: int = 10) -> (List[Tag], int):
|
||||
""" 获取所有的标签 """
|
||||
result = TagDao.search_tags(keyword, page, limit)
|
||||
return result, TagDao.count_tags(keyword)
|
||||
|
||||
@classmethod
|
||||
def create_tag(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
name: str) -> Tag:
|
||||
# 查询是否有重名的标签名称
|
||||
exist_tag = TagDao.get_tag_by_name(name)
|
||||
if exist_tag:
|
||||
raise TagExistError.http_exception()
|
||||
new_tag = Tag(name=name, user_id=login_user.user_id)
|
||||
new_tag = TagDao.insert_tag(new_tag)
|
||||
return new_tag
|
||||
|
||||
@classmethod
|
||||
def update_tag(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
tag_id: int,
|
||||
name: str) -> Tag:
|
||||
tag_info = TagDao.get_tag_by_id(tag_id)
|
||||
if not tag_info:
|
||||
raise TagNotExistError.http_exception()
|
||||
# 查询是否有重名的标签名称
|
||||
exist_tag = TagDao.get_tag_by_name(name)
|
||||
if exist_tag and exist_tag.id != tag_id:
|
||||
raise TagExistError.http_exception()
|
||||
|
||||
tag_info.name = name
|
||||
new_tag = TagDao.insert_tag(tag_info)
|
||||
return new_tag
|
||||
|
||||
@classmethod
|
||||
def delete_tag(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
tag_id: int) -> bool:
|
||||
""" 删除标签 """
|
||||
return TagDao.delete_tag(tag_id)
|
||||
|
||||
@classmethod
|
||||
def check_tag_link_permission(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
resource_id: str,
|
||||
resource_type: ResourceTypeEnum) -> bool:
|
||||
""" 检查是否允许给资源打标签 """
|
||||
if login_user.is_admin():
|
||||
return True
|
||||
resource_info = None
|
||||
if resource_type == ResourceTypeEnum.ASSISTANT:
|
||||
resource_info = AssistantDao.get_one_assistant(resource_id)
|
||||
elif resource_type == ResourceTypeEnum.FLOW:
|
||||
resource_info = FlowDao.get_flow_by_id(resource_id)
|
||||
elif resource_type == ResourceTypeEnum.WORK_FLOW:
|
||||
resource_info = FlowDao.get_flow_by_id(resource_id)
|
||||
else:
|
||||
raise HTTPException(status_code=404, detail="资源类型不支持")
|
||||
if not resource_info:
|
||||
raise HTTPException(status_code=404, detail="资源不存在")
|
||||
# 是资源的创建人
|
||||
if resource_info.user_id == login_user.user_id:
|
||||
return True
|
||||
|
||||
# 获取资源所属的用户组
|
||||
resource_groups = GroupResourceDao.get_resource_group(resource_type, resource_id)
|
||||
resource_groups = [int(one.group_id) for one in resource_groups]
|
||||
# 判断下操作人是否是用户组的管理员
|
||||
if not login_user.check_groups_admin(resource_groups):
|
||||
raise UnAuthorizedError.http_exception()
|
||||
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def create_tag_link(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
tag_id: int,
|
||||
resource_id: str,
|
||||
resource_type: ResourceTypeEnum) -> TagLink:
|
||||
""" 建立资源和标签的关联 """
|
||||
cls.check_tag_link_permission(request, login_user, resource_id, resource_type)
|
||||
|
||||
new_link = TagLink(tag_id=tag_id, resource_id=resource_id, resource_type=resource_type.value,
|
||||
user_id=login_user.user_id)
|
||||
try:
|
||||
new_link = TagDao.insert_tag_link(new_link)
|
||||
except Exception as e:
|
||||
logger.error(f'tag_link_error: {e}')
|
||||
raise TagExistError.http_exception()
|
||||
return new_link
|
||||
|
||||
@classmethod
|
||||
def delete_tag_link(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
tag_id: int,
|
||||
resource_id: str,
|
||||
resource_type: ResourceTypeEnum) -> bool:
|
||||
""" 删除资源和标签的关联 """
|
||||
cls.check_tag_link_permission(request, login_user, resource_id, resource_type)
|
||||
|
||||
return TagDao.delete_resource_tag(tag_id, resource_id, resource_type)
|
||||
|
||||
@classmethod
|
||||
def get_home_tag(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload) -> List[Tag]:
|
||||
""" 获取首页展示的标签列表 """
|
||||
home_tags = ConfigDao.get_config(ConfigKeyEnum.HOME_TAGS)
|
||||
if not home_tags:
|
||||
return []
|
||||
home_tags = json.loads(home_tags.value)
|
||||
tags = TagDao.get_tags_by_ids(home_tags)
|
||||
|
||||
tags = sorted(tags, key=lambda x: home_tags.index(x.id))
|
||||
return tags
|
||||
|
||||
@classmethod
|
||||
def update_home_tag(cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
tag_ids: List[int]) -> bool:
|
||||
""" 更新首页展示的标签列表 """
|
||||
home_tags = ConfigDao.get_config(ConfigKeyEnum.HOME_TAGS)
|
||||
if not home_tags:
|
||||
home_tags = Config(key=ConfigKeyEnum.HOME_TAGS.value, value=json.dumps(tag_ids))
|
||||
else:
|
||||
home_tags.value = json.dumps(tag_ids)
|
||||
|
||||
ConfigDao.insert_config(home_tags)
|
||||
return True
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1 @@
|
||||
from .tool import ToolServices
|
||||
@@ -0,0 +1,106 @@
|
||||
import json
|
||||
from typing import Optional, Type
|
||||
|
||||
from langchain_core.tools import BaseTool
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from bisheng.api.services.knowledge_imp import decide_vectorstores
|
||||
from bisheng.database.models.knowledge import KnowledgeDao
|
||||
from bisheng.database.models.linsight_session_version import LinsightSessionVersionDao
|
||||
from bisheng.database.models.llm_server import LLMDao
|
||||
from bisheng.interface.importing.utils import import_vectorstore
|
||||
from bisheng.interface.initialize.loading import instantiate_vectorstore
|
||||
from bisheng.utils.embedding import decide_embeddings
|
||||
|
||||
|
||||
class ToolInput(BaseModel):
|
||||
query: str = Field(..., description='需要检索的关键词')
|
||||
knowledge_id: Optional[str] = Field(default=None, description='语义检索库id')
|
||||
limit: Optional[int] = Field(default=2, description='返回结果的最大数量')
|
||||
call_reason: str = Field(default='', description='调用该工具的原因,原因中不要使用id来描述文件或知识库')
|
||||
|
||||
|
||||
class SearchKnowledgeBase(BaseTool):
|
||||
name: str = "search_knowledge_base"
|
||||
description: str = """在语义检索库中搜索相关内容。
|
||||
|
||||
用法:在你需要在知识库中进行语义搜索时,调用此工具。
|
||||
|
||||
Args:
|
||||
query: 需要检索的关键词
|
||||
knowledge_id: 语义检索库id
|
||||
limit: 返回结果的最大数量,默认为2
|
||||
|
||||
Returns:
|
||||
包含搜索结果(chunk的列表)的字典"""
|
||||
args_schema: Type[BaseModel] = ToolInput
|
||||
|
||||
def _run(self, query: str, knowledge_id: Optional[str] = None,
|
||||
**kwargs) -> str:
|
||||
"""Use the tool."""
|
||||
return "not supported in sync mode, please use async version"
|
||||
|
||||
async def _arun(self, query: str, knowledge_id: Optional[str] = None,
|
||||
**kwargs) -> str:
|
||||
limit = kwargs.get('limit', None) or 2
|
||||
if not query:
|
||||
raise ValueError("query 参数不能为空")
|
||||
|
||||
try:
|
||||
knowledge_id = int(knowledge_id)
|
||||
return await self.search_knowledge(query, knowledge_id, limit)
|
||||
except ValueError:
|
||||
return await self.search_linsight_file(query, knowledge_id, limit)
|
||||
|
||||
async def base_search(self, vector_client, query: str, k: int):
|
||||
documents = await vector_client.asimilarity_search(query, k=k)
|
||||
if not documents:
|
||||
# "没有找到相关的知识内容"
|
||||
return '{"状态": "无结果", "错误信息":"没有找到相关的知识内容"}'
|
||||
result = {
|
||||
"状态": "成功",
|
||||
"结果": [one.page_content for one in documents]
|
||||
}
|
||||
result = json.dumps(result, ensure_ascii=False, indent=2)
|
||||
|
||||
return result
|
||||
|
||||
async def search_linsight_file(self, query: str, file_id: str, limit: int) -> str:
|
||||
"""检索Linsight用户上传的文件"""
|
||||
session_info = await LinsightSessionVersionDao.get_session_version_by_file_id(file_id=file_id)
|
||||
if not session_info:
|
||||
raise Exception("文件不存在或已被删除")
|
||||
files = session_info.files
|
||||
file_info = None
|
||||
for one in files:
|
||||
if one.get("file_id") == file_id:
|
||||
file_info = one
|
||||
break
|
||||
if not file_info:
|
||||
raise Exception("文件不存在或已被删除")
|
||||
class_obj = import_vectorstore('Milvus')
|
||||
embeddings = decide_embeddings(file_info.get("embedding_model_id"))
|
||||
params = {
|
||||
'collection_name': file_info.get("collection_name"),
|
||||
'embedding': embeddings,
|
||||
'metadata_expr': f'file_id in {[file_id]}'
|
||||
}
|
||||
milvus_client = instantiate_vectorstore('Milvus', class_object=class_obj, params=params)
|
||||
return await self.base_search(milvus_client, query, limit)
|
||||
|
||||
async def search_knowledge(self, query: str, knowledge_id: int, limit: int) -> str:
|
||||
knowledge_info = KnowledgeDao.query_by_id(knowledge_id)
|
||||
if not knowledge_info:
|
||||
raise Exception("知识库不存在或已被删除")
|
||||
if not knowledge_info.model:
|
||||
# "知识库未配置embedding模型"
|
||||
raise Exception("知识库未配置embedding模型")
|
||||
embed_info = LLMDao.get_model_by_id(int(knowledge_info.model))
|
||||
if not embed_info:
|
||||
# "知识库配置的embedding模型不存在或已被删除"
|
||||
raise Exception("知识库配置的embedding模型不存在或已被删除")
|
||||
embeddings = decide_embeddings(knowledge_info.model)
|
||||
milvus_client = decide_vectorstores(
|
||||
knowledge_info.collection_name, "Milvus", embeddings
|
||||
)
|
||||
return await self.base_search(milvus_client, query, limit)
|
||||
@@ -0,0 +1,332 @@
|
||||
import json
|
||||
from typing import Optional, List
|
||||
|
||||
import yaml
|
||||
from fastapi import Request
|
||||
from langchain_core.tools import BaseTool
|
||||
from loguru import logger
|
||||
from pydantic import BaseModel, ConfigDict
|
||||
|
||||
from bisheng.api.errcode.assistant import ToolTypeNotExistsError, ToolTypeRepeatError
|
||||
from bisheng.api.errcode.base import ServerError, UnAuthorizedError
|
||||
from bisheng.api.services.openapi import OpenApiSchema
|
||||
from bisheng.api.services.tool.langchain_tool.search_knowledge import SearchKnowledgeBase
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.utils import get_url_content
|
||||
from bisheng.database.constants import ToolPresetType
|
||||
from bisheng.database.models.gpts_tools import GptsToolsDao, GptsTools, GptsToolsType, GptsToolsTypeRead
|
||||
from bisheng.database.models.role_access import AccessType
|
||||
from bisheng.mcp_manage.manager import ClientManager
|
||||
from bisheng.utils import md5_hash
|
||||
from bisheng_langchain.gpts.load_tools import load_tools
|
||||
|
||||
|
||||
class ToolServices(BaseModel):
|
||||
""" 工具服务类 """
|
||||
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||
|
||||
request: Optional[Request] = None
|
||||
login_user: Optional[UserPayload] = None
|
||||
|
||||
async def parse_openapi_schema(self, download_url: str, file_content: str) -> GptsToolsTypeRead:
|
||||
if download_url:
|
||||
try:
|
||||
file_content = await get_url_content(download_url)
|
||||
except Exception as e:
|
||||
logger.exception(f'file {download_url} download error')
|
||||
raise ServerError.http_exception(msg='url文件下载失败:' + str(e))
|
||||
if not file_content:
|
||||
raise ServerError.http_exception(msg='schema内容不能为空')
|
||||
# 根据文件内容是否以`{`开头判断用什么解析方式
|
||||
try:
|
||||
if file_content.startswith('{'):
|
||||
res = json.loads(file_content)
|
||||
else:
|
||||
res = yaml.safe_load(file_content)
|
||||
except Exception as e:
|
||||
logger.exception(f'openapi schema parse error {e}')
|
||||
raise ServerError.http_exception(msg=f'openapi schema解析报错,请检查内容是否符合json或者yaml格式: {str(e)}')
|
||||
|
||||
# 解析openapi schema转为助手工具的格式
|
||||
try:
|
||||
schema = OpenApiSchema(res)
|
||||
schema.parse_server()
|
||||
if not schema.default_server.startswith(('http', 'https')):
|
||||
raise ServerError.http_exception(msg=f'server中的url必须以http或者https开头: {schema.default_server}')
|
||||
tool_type = GptsToolsTypeRead(name=schema.title,
|
||||
description=schema.description,
|
||||
is_preset=ToolPresetType.API.value,
|
||||
server_host=schema.default_server,
|
||||
openapi_schema=file_content,
|
||||
api_location=schema.api_location,
|
||||
parameter_name=schema.parameter_name,
|
||||
auth_type=schema.auth_type,
|
||||
auth_method=schema.auth_method,
|
||||
children=[])
|
||||
# 解析获取所有的api
|
||||
schema.parse_paths()
|
||||
for one in schema.apis:
|
||||
tool_type.children.append(
|
||||
GptsTools(
|
||||
name=one['operationId'],
|
||||
desc=one['description'],
|
||||
tool_key=md5_hash(one['operationId']),
|
||||
is_preset=0,
|
||||
is_delete=0,
|
||||
api_params=one['parameters'],
|
||||
extra=json.dumps(one, ensure_ascii=False),
|
||||
))
|
||||
return tool_type
|
||||
except Exception as e:
|
||||
logger.exception(f'openapi schema parse error {e}')
|
||||
raise ServerError.http_exception(msg='openapi schema解析失败:' + str(e))
|
||||
|
||||
async def parse_mcp_schema(self, file_content: str) -> GptsToolsTypeRead:
|
||||
try:
|
||||
result = json.loads(file_content)
|
||||
mcp_servers = result['mcpServers']
|
||||
except Exception as e:
|
||||
logger.exception(f'mcp tool schema parse error {e}')
|
||||
raise ServerError.http_exception(msg=f'mcp工具配置解析失败,请检查内容是否符合mcp配置格式: {str(e)}')
|
||||
tool_type = None
|
||||
for key, value in mcp_servers.items():
|
||||
# 解析mcp服务配置
|
||||
tool_type = GptsToolsTypeRead(name=value.get('name', ''),
|
||||
server_host=value.get('url', ''),
|
||||
description=value.get('description', ''),
|
||||
is_preset=ToolPresetType.MCP.value,
|
||||
openapi_schema=file_content,
|
||||
children=[])
|
||||
# 实例化mcp服务对象,获取工具列表
|
||||
client = await ClientManager.connect_mcp_from_json(result)
|
||||
|
||||
tools = await client.list_tools()
|
||||
|
||||
for one in tools:
|
||||
tool_type.children.append(GptsTools(
|
||||
name=one.name,
|
||||
desc=one.description,
|
||||
tool_key=md5_hash(one.name),
|
||||
is_preset=ToolPresetType.MCP.value,
|
||||
api_params=ToolServices.convert_input_schema(one.inputSchema),
|
||||
extra=one.model_dump_json(),
|
||||
))
|
||||
break
|
||||
if tool_type is None:
|
||||
raise ServerError.http_exception(msg='mcp服务配置解析失败,请检查配置里是否配置了mcpServers')
|
||||
return tool_type
|
||||
|
||||
@classmethod
|
||||
async def _update_gpts_tools(cls, exist_tool_type: GptsToolsType, req: GptsToolsTypeRead) -> GptsToolsTypeRead:
|
||||
exist_tool_type.name = req.name
|
||||
exist_tool_type.logo = req.logo
|
||||
exist_tool_type.description = req.description
|
||||
exist_tool_type.server_host = req.server_host
|
||||
exist_tool_type.auth_method = req.auth_method
|
||||
exist_tool_type.api_key = req.api_key
|
||||
exist_tool_type.auth_type = req.auth_type
|
||||
exist_tool_type.openapi_schema = req.openapi_schema
|
||||
tool_extra = {"api_location": req.api_location, "parameter_name": req.parameter_name}
|
||||
exist_tool_type.extra = json.dumps(tool_extra, ensure_ascii=False)
|
||||
|
||||
children_map = {}
|
||||
for one in req.children:
|
||||
children_map[one.name] = one
|
||||
|
||||
# 获取此类别下旧的API列表
|
||||
old_tool_list = GptsToolsDao.get_list_by_type([exist_tool_type.id])
|
||||
# 需要被删除的工具列表
|
||||
delete_tool_id_list = []
|
||||
# 需要被更新的工具列表
|
||||
update_tool_list = []
|
||||
for one in old_tool_list:
|
||||
# 说明此工具 需要删除
|
||||
if children_map.get(one.name) is None:
|
||||
delete_tool_id_list.append(one.id)
|
||||
else:
|
||||
# 说明此工具需要更新
|
||||
new_tool_info = children_map.pop(one.name)
|
||||
one.name = new_tool_info.name
|
||||
one.desc = new_tool_info.desc
|
||||
one.extra = new_tool_info.extra
|
||||
one.api_params = new_tool_info.api_params
|
||||
update_tool_list.append(one)
|
||||
|
||||
add_children = []
|
||||
for one in children_map.values():
|
||||
one.id = None
|
||||
one.user_id = exist_tool_type.user_id
|
||||
one.is_preset = exist_tool_type.is_preset
|
||||
one.is_delete = 0
|
||||
add_children.append(one)
|
||||
|
||||
GptsToolsDao.update_tool_type(exist_tool_type, delete_tool_id_list,
|
||||
add_children, update_tool_list)
|
||||
|
||||
children = GptsToolsDao.get_list_by_type([exist_tool_type.id])
|
||||
return GptsToolsTypeRead(**exist_tool_type.model_dump(), children=children)
|
||||
|
||||
@classmethod
|
||||
async def update_gpts_tools(cls, user: UserPayload, req: GptsToolsTypeRead) -> GptsToolsTypeRead:
|
||||
"""
|
||||
更新工具类别,包括更新工具类别的名称和删除、新增工具类别的API
|
||||
"""
|
||||
# 尝试解析下openapi schema看下是否可以正常解析, 不能的话保存不允许保存
|
||||
tool_service = ToolServices()
|
||||
if req.is_preset == ToolPresetType.API.value:
|
||||
await tool_service.parse_openapi_schema('', req.openapi_schema)
|
||||
elif req.is_preset == ToolPresetType.MCP.value:
|
||||
await tool_service.parse_mcp_schema(req.openapi_schema)
|
||||
|
||||
exist_tool_type = GptsToolsDao.get_one_tool_type(req.id)
|
||||
if not exist_tool_type:
|
||||
raise ToolTypeNotExistsError.http_exception()
|
||||
if req.name.__len__() > 1000 or req.name.__len__() == 0:
|
||||
raise ServerError.http_exception(msg="名字不符合规范:至少1个字符,不能超过1000个字符")
|
||||
|
||||
# 判断工具类别名称是否重复
|
||||
tool_type = GptsToolsDao.get_one_tool_type_by_name(user.user_id, req.name)
|
||||
if tool_type and tool_type.id != exist_tool_type.id:
|
||||
raise ToolTypeRepeatError.http_exception()
|
||||
# 判断是否有更新权限
|
||||
if not user.access_check(exist_tool_type.user_id, str(exist_tool_type.id), AccessType.GPTS_TOOL_WRITE):
|
||||
raise UnAuthorizedError.http_exception()
|
||||
|
||||
return await cls._update_gpts_tools(exist_tool_type, req)
|
||||
|
||||
async def refresh_all_mcp(self) -> str:
|
||||
""" return mcp server error msg """
|
||||
# get user all mcp tool
|
||||
tool_types = GptsToolsDao.get_user_tool_type(self.login_user.user_id, is_preset=ToolPresetType.MCP)
|
||||
if not tool_types:
|
||||
return ''
|
||||
|
||||
tools = GptsToolsDao.get_list_by_type(tool_type_ids=[one.id for one in tool_types])
|
||||
tools_map = {}
|
||||
for one in tools:
|
||||
if one.type not in tools_map:
|
||||
tools_map[one.type] = []
|
||||
tools_map[one.type].append(one)
|
||||
error_msg = ''
|
||||
for one in tool_types:
|
||||
try:
|
||||
await self.refresh_mcp_tools(one, tools_map.get(one.id, []))
|
||||
except Exception as e:
|
||||
logger.exception(f'{one.name}刷新工具失败:')
|
||||
error_msg += f'{one.name}工具获取失败,请重试\n'
|
||||
return error_msg
|
||||
|
||||
async def refresh_mcp_tools(self, tool_type: GptsToolsType, old_tools: list[GptsTools]):
|
||||
""" refresh mcp tools """
|
||||
# 1. get all new tools
|
||||
# 实例化mcp服务对象,获取工具列表
|
||||
client = await ClientManager.connect_mcp_from_json(tool_type.openapi_schema)
|
||||
tools = await client.list_tools()
|
||||
children = []
|
||||
for one in tools:
|
||||
children.append(GptsTools(
|
||||
name=one.name,
|
||||
desc=one.description,
|
||||
is_preset=ToolPresetType.MCP.value,
|
||||
api_params=self.convert_input_schema(one.inputSchema),
|
||||
extra=one.model_dump_json(),
|
||||
type=tool_type.id,
|
||||
))
|
||||
|
||||
req = GptsToolsTypeRead(**tool_type.model_dump(), children=children)
|
||||
await self._update_gpts_tools(tool_type, req)
|
||||
|
||||
@classmethod
|
||||
def convert_input_schema(cls, input_schema: dict):
|
||||
""" 转换mcp工具的输入参数 为自定义工具的格式"""
|
||||
required = input_schema.get('required', [])
|
||||
properties = input_schema.get('properties', {})
|
||||
res = []
|
||||
for filed, field_info in properties.items():
|
||||
res.append({
|
||||
'in': "query",
|
||||
'name': filed,
|
||||
'description': field_info.get('description'),
|
||||
'required': filed in required,
|
||||
'schema': {
|
||||
'type': field_info.get('type'),
|
||||
}
|
||||
})
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
async def init_linsight_tools(cls, root_path: str) -> List[BaseTool]:
|
||||
""" 初始化Linsight 默认的工具, 特殊点在于本地文件工具初始化的参数不是固定的,而是再运行期间确定的 """
|
||||
# 加载本地文件操作相关工具
|
||||
local_file_tools = load_tools({
|
||||
"list_files": {"root_path": root_path},
|
||||
"get_file_details": {"root_path": root_path},
|
||||
"search_files": {"root_path": root_path},
|
||||
# "search_text_in_file": {"root_path": root_path},
|
||||
"read_text_file": {"root_path": root_path},
|
||||
"add_text_to_file": {"root_path": root_path},
|
||||
"replace_file_lines": {"root_path": root_path},
|
||||
})
|
||||
knowledge_tools = [SearchKnowledgeBase()]
|
||||
return knowledge_tools + local_file_tools
|
||||
|
||||
@classmethod
|
||||
async def get_linsight_tools(cls) -> list[GptsToolsTypeRead]:
|
||||
return [
|
||||
GptsToolsTypeRead(
|
||||
id=100000,
|
||||
name="知识库和文件内容检索",
|
||||
description="检索组织知识库、个人知识库以及本地上传文件的内容",
|
||||
children=[
|
||||
GptsTools(
|
||||
id=100001,
|
||||
name="知识库和文件内容检索",
|
||||
desc="检索组织知识库、个人知识库以及本地上传文件的内容。",
|
||||
tool_key="search_knowledge_base",
|
||||
)
|
||||
]
|
||||
),
|
||||
GptsToolsTypeRead(
|
||||
id=200000,
|
||||
name="文件操作",
|
||||
description="本地文件系统的浏览、搜索与编辑工具集",
|
||||
children=[
|
||||
GptsTools(
|
||||
id=200001,
|
||||
name="获取所有文件和目录",
|
||||
desc="列出指定目录下的所有文件和子目录。",
|
||||
tool_key="list_files"
|
||||
),
|
||||
GptsTools(
|
||||
id=200002,
|
||||
name="获取文件详细信息",
|
||||
desc="获取指定文件的文件名、文件大小、文件地址、字数、行数等详细信息。",
|
||||
tool_key="get_file_details"
|
||||
),
|
||||
GptsTools(
|
||||
id=200003,
|
||||
name="搜索文件",
|
||||
desc="在指定目录中搜索文件和子目录。",
|
||||
tool_key="search_files"
|
||||
),
|
||||
GptsTools(
|
||||
id=200004,
|
||||
name="读取文件内容",
|
||||
desc="读取本地文本文件的内容。",
|
||||
tool_key="read_text_file"
|
||||
),
|
||||
GptsTools(
|
||||
id=200005,
|
||||
name="写入文件内容",
|
||||
desc="将文本内容追加到文本文件,如果文件不存在,则创建文件",
|
||||
tool_key="add_text_to_file"
|
||||
),
|
||||
GptsTools(
|
||||
id=200006,
|
||||
name="替换文件指定行范围内容",
|
||||
desc="替换文件中的指定行范围。",
|
||||
tool_key="replace_file_lines"
|
||||
),
|
||||
]
|
||||
)
|
||||
]
|
||||
@@ -1,8 +1,29 @@
|
||||
import functools
|
||||
import json
|
||||
from base64 import b64decode
|
||||
from typing import List, Dict
|
||||
|
||||
import rsa
|
||||
from bisheng.api.errcode.base import UnAuthorizedError
|
||||
from bisheng.api.errcode.user import (UserLoginOfflineError, UserNameAlreadyExistError,
|
||||
UserNeedGroupAndRoleError)
|
||||
from bisheng.api.JWT import ACCESS_TOKEN_EXPIRE_TIME
|
||||
from bisheng.api.utils import md5_hash
|
||||
from bisheng.api.v1.schemas import CreateUserReq
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.database.constants import AdminRole
|
||||
from bisheng.database.models.assistant import Assistant, AssistantDao
|
||||
from bisheng.database.models.flow import Flow, FlowDao, FlowRead
|
||||
from bisheng.database.models.group import GroupDao
|
||||
from bisheng.database.models.knowledge import Knowledge, KnowledgeDao, KnowledgeRead
|
||||
from bisheng.database.models.role_access import AccessType, RoleAccessDao
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.database.models.user import User, UserDao
|
||||
from bisheng.database.models.user_group import UserGroupDao
|
||||
from bisheng.database.models.user_role import UserRoleDao
|
||||
from bisheng.settings import settings
|
||||
from bisheng.utils.constants import RSA_KEY, USER_CURRENT_SESSION
|
||||
from fastapi import Depends, HTTPException, Request
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
|
||||
|
||||
class UserPayload:
|
||||
@@ -10,13 +31,41 @@ class UserPayload:
|
||||
def __init__(self, **kwargs):
|
||||
self.user_id = kwargs.get('user_id')
|
||||
self.user_role = kwargs.get('role')
|
||||
self.group_cache = {}
|
||||
if self.user_role != 'admin': # 非管理员用户,需要获取他的角色列表
|
||||
roles = UserRoleDao.get_user_roles(self.user_id)
|
||||
self.user_role = [one.role_id for one in roles]
|
||||
self.user_name = kwargs.get('user_name')
|
||||
|
||||
def is_admin(self):
|
||||
return self.user_role == 'admin'
|
||||
|
||||
def access_check(self, owner_user_id: int, target_id: str, access_type: AccessType) -> bool:
|
||||
if self.is_admin():
|
||||
if self.user_role == 'admin':
|
||||
return True
|
||||
if isinstance(self.user_role, list):
|
||||
for one in self.user_role:
|
||||
if one == AdminRole:
|
||||
return True
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def wrapper_access_check(func):
|
||||
"""
|
||||
权限检查的装饰器
|
||||
如果是admin用户则不执行后续具体的检查逻辑
|
||||
"""
|
||||
|
||||
@functools.wraps(func)
|
||||
def wrapper(*args, **kwargs):
|
||||
if args[0].is_admin():
|
||||
return True
|
||||
return func(*args, **kwargs)
|
||||
|
||||
return wrapper
|
||||
|
||||
@wrapper_access_check
|
||||
def access_check(self, owner_user_id: int, target_id: str, access_type: AccessType) -> bool:
|
||||
"""
|
||||
检查用户是否有某个资源的权限
|
||||
"""
|
||||
# 判断是否属于本人资源
|
||||
if self.user_id == owner_user_id:
|
||||
return True
|
||||
@@ -25,9 +74,139 @@ class UserPayload:
|
||||
return True
|
||||
return False
|
||||
|
||||
@wrapper_access_check
|
||||
def copiable_check(self, owner_user_id: int) -> bool:
|
||||
"""
|
||||
检查用户是否有某个资源复制权限
|
||||
"""
|
||||
# 判断是否属于本人资源
|
||||
if self.user_id == owner_user_id:
|
||||
return True
|
||||
return False
|
||||
|
||||
@wrapper_access_check
|
||||
def check_group_admin(self, group_id: int) -> bool:
|
||||
"""
|
||||
检查用户是否是某个组的管理员
|
||||
"""
|
||||
# 判断是否是用户组的管理员
|
||||
user_group = UserGroupDao.get_user_admin_group(self.user_id)
|
||||
if not user_group:
|
||||
return False
|
||||
for one in user_group:
|
||||
if one.group_id == group_id:
|
||||
return True
|
||||
return False
|
||||
|
||||
@wrapper_access_check
|
||||
def check_groups_admin(self, group_ids: List[int]) -> bool:
|
||||
"""
|
||||
检查用户是否是用户组列表中的管理员,有一个就是true
|
||||
"""
|
||||
user_groups = UserGroupDao.get_user_admin_group(self.user_id)
|
||||
for one in user_groups:
|
||||
if one.is_group_admin and one.group_id in group_ids:
|
||||
return True
|
||||
return False
|
||||
|
||||
def get_user_groups(self, user_id: int) -> List[Dict]:
|
||||
""" 查询用户的角色列表 """
|
||||
user_groups = UserGroupDao.get_user_group(user_id)
|
||||
user_group_ids: List[int] = [one_group.group_id for one_group in user_groups]
|
||||
res = []
|
||||
for i in range(len(user_group_ids) - 1, -1, -1):
|
||||
if self.group_cache.get(user_group_ids[i]):
|
||||
res.append(self.group_cache.get(user_group_ids[i]))
|
||||
del user_group_ids[i]
|
||||
# 将没有缓存的角色信息查询数据库
|
||||
if user_group_ids:
|
||||
group_list = GroupDao.get_group_by_ids(user_group_ids)
|
||||
for group_info in group_list:
|
||||
self.group_cache[group_info.id] = {'id': group_info.id, 'name': group_info.group_name}
|
||||
res.append(self.group_cache.get(group_info.id))
|
||||
return res
|
||||
|
||||
class UserService:
|
||||
|
||||
@classmethod
|
||||
def decrypt_md5_password(cls, password: str):
|
||||
if value := redis_client.get(RSA_KEY):
|
||||
private_key = value[1]
|
||||
password = md5_hash(rsa.decrypt(b64decode(password), private_key).decode('utf-8'))
|
||||
else:
|
||||
password = md5_hash(password)
|
||||
return password
|
||||
|
||||
@classmethod
|
||||
def create_user(cls, request: Request, login_user: UserPayload, req_data: CreateUserReq):
|
||||
"""
|
||||
创建用户
|
||||
"""
|
||||
exists_user = UserDao.get_user_by_username(req_data.user_name)
|
||||
if exists_user:
|
||||
# 抛出异常
|
||||
raise UserNameAlreadyExistError.http_exception()
|
||||
user = User(
|
||||
user_name=req_data.user_name,
|
||||
password=cls.decrypt_md5_password(req_data.password),
|
||||
)
|
||||
group_ids = []
|
||||
role_ids = []
|
||||
for one in req_data.group_roles:
|
||||
group_ids.append(one.group_id)
|
||||
role_ids.extend(one.role_ids)
|
||||
if not group_ids or not role_ids:
|
||||
raise UserNeedGroupAndRoleError.http_exception()
|
||||
user = UserDao.add_user_with_groups_and_roles(user, group_ids, role_ids)
|
||||
return user
|
||||
|
||||
|
||||
def sso_login():
|
||||
pass
|
||||
|
||||
|
||||
def gen_user_role(db_user: User):
|
||||
# 查询用户的角色列表
|
||||
db_user_role = UserRoleDao.get_user_roles(db_user.user_id)
|
||||
role = ''
|
||||
role_ids = []
|
||||
for user_role in db_user_role:
|
||||
if user_role.role_id == 1:
|
||||
# 是管理员,忽略其他的角色
|
||||
role = 'admin'
|
||||
else:
|
||||
role_ids.append(user_role.role_id)
|
||||
if role != 'admin':
|
||||
# 判断是否是用户组管理员
|
||||
db_user_groups = UserGroupDao.get_user_admin_group(db_user.user_id)
|
||||
if len(db_user_groups) > 0:
|
||||
role = 'group_admin'
|
||||
else:
|
||||
role = role_ids
|
||||
# 获取用户的菜单栏权限列表
|
||||
web_menu = RoleAccessDao.get_role_access(role_ids, AccessType.WEB_MENU)
|
||||
web_menu = list(set([one.third_id for one in web_menu]))
|
||||
return role, web_menu
|
||||
|
||||
|
||||
def gen_user_jwt(db_user: User):
|
||||
if 1 == db_user.delete:
|
||||
raise HTTPException(status_code=500, detail='该账号已被禁用,请联系管理员')
|
||||
# 查询角色
|
||||
role, web_menu = gen_user_role(db_user)
|
||||
# 生成JWT令牌
|
||||
payload = {'user_name': db_user.user_name, 'user_id': db_user.user_id, 'role': role}
|
||||
# Create the tokens and passing to set_access_cookies or set_refresh_cookies
|
||||
access_token = AuthJWT().create_access_token(subject=json.dumps(payload),
|
||||
expires_time=ACCESS_TOKEN_EXPIRE_TIME)
|
||||
|
||||
refresh_token = AuthJWT().create_refresh_token(subject=db_user.user_name)
|
||||
|
||||
# Set the JWT cookies in the response
|
||||
return access_token, refresh_token, role, web_menu
|
||||
|
||||
|
||||
def get_knowledge_list_by_access(role_id: int, name: str, page_num: int, page_size: int):
|
||||
|
||||
count_filter = []
|
||||
if name:
|
||||
count_filter.append(Knowledge.name.like('%{}%'.format(name)))
|
||||
@@ -55,7 +234,6 @@ def get_knowledge_list_by_access(role_id: int, name: str, page_num: int, page_si
|
||||
|
||||
|
||||
def get_flow_list_by_access(role_id: int, name: str, page_num: int, page_size: int):
|
||||
|
||||
count_filter = []
|
||||
if name:
|
||||
count_filter.append(Flow.name.like('%{}%'.format(name)))
|
||||
@@ -83,7 +261,6 @@ def get_flow_list_by_access(role_id: int, name: str, page_num: int, page_size: i
|
||||
|
||||
|
||||
def get_assistant_list_by_access(role_id: int, name: str, page_num: int, page_size: int):
|
||||
|
||||
count_filter = []
|
||||
if name:
|
||||
count_filter.append(Assistant.name.like('%{}%'.format(name)))
|
||||
@@ -106,3 +283,33 @@ def get_assistant_list_by_access(role_id: int, name: str, page_num: int, page_si
|
||||
'total':
|
||||
total_count
|
||||
}
|
||||
|
||||
|
||||
async def get_login_user(authorize: AuthJWT = Depends()) -> UserPayload:
|
||||
"""
|
||||
获取当前登录的用户
|
||||
"""
|
||||
# 校验是否过期,过期则直接返回http 状态码的 401
|
||||
authorize.jwt_required()
|
||||
|
||||
current_user = json.loads(authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
|
||||
# 判断是否允许多点登录
|
||||
if not settings.get_system_login_method().allow_multi_login:
|
||||
# 获取access_token
|
||||
current_token = redis_client.get(USER_CURRENT_SESSION.format(user.user_id))
|
||||
# 登录被挤下线了,http状态码是200, status_code是特殊code
|
||||
if current_token != authorize._token:
|
||||
raise UserLoginOfflineError.http_exception()
|
||||
return user
|
||||
|
||||
|
||||
async def get_admin_user(authorize: AuthJWT = Depends()) -> UserPayload:
|
||||
"""
|
||||
获取超级管理账号,非超级管理员用户,抛出异常
|
||||
"""
|
||||
login_user = await get_login_user(authorize)
|
||||
if not login_user.is_admin():
|
||||
raise UnAuthorizedError.http_exception()
|
||||
return login_user
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from bisheng.template.field.base import TemplateField
|
||||
from bisheng.template.template.base import Template
|
||||
from langchain.pydantic_v1 import BaseModel
|
||||
from pydantic import BaseModel
|
||||
from langchain_core.language_models import BaseLanguageModel
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,312 @@
|
||||
from typing import Dict, Optional
|
||||
|
||||
from bisheng.utils import generate_uuid
|
||||
from fastapi.encoders import jsonable_encoder
|
||||
from langchain.memory import ConversationBufferWindowMemory
|
||||
|
||||
from bisheng.api.errcode.base import NotFoundError, UnAuthorizedError
|
||||
from bisheng.api.errcode.flow import WorkFlowInitError
|
||||
from bisheng.api.services.base import BaseService
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.v1.schemas import ChatResponse
|
||||
from bisheng.api.v1.schema.workflow import WorkflowEvent, WorkflowEventType, WorkflowInputSchema, WorkflowInputItem, \
|
||||
WorkflowOutputSchema
|
||||
from bisheng.chat.utils import SourceType
|
||||
from bisheng.database.models.flow import FlowDao, FlowType, FlowStatus
|
||||
from bisheng.database.models.flow_version import FlowVersionDao
|
||||
from bisheng.database.models.group_resource import GroupResourceDao, ResourceTypeEnum
|
||||
from bisheng.database.models.role_access import AccessType, RoleAccessDao
|
||||
from bisheng.database.models.tag import TagDao
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.database.models.user_role import UserRoleDao
|
||||
from bisheng.workflow.callback.base_callback import BaseCallback
|
||||
from bisheng.workflow.common.node import BaseNodeData, NodeType
|
||||
from bisheng.workflow.graph.graph_state import GraphState
|
||||
from bisheng.workflow.graph.workflow import Workflow
|
||||
from bisheng.workflow.nodes.node_manage import NodeFactory
|
||||
|
||||
|
||||
class WorkFlowService(BaseService):
|
||||
|
||||
@classmethod
|
||||
def get_all_flows(cls, user: UserPayload, name: str, status: int, tag_id: Optional[int], flow_type: Optional[int],
|
||||
page: int = 1,
|
||||
page_size: int = 10) -> (list[dict], int):
|
||||
"""
|
||||
获取所有技能
|
||||
"""
|
||||
# 通过tag获取id列表
|
||||
flow_ids = []
|
||||
if tag_id:
|
||||
ret = TagDao.get_resources_by_tags_batch([tag_id], [ResourceTypeEnum.FLOW, ResourceTypeEnum.WORK_FLOW,
|
||||
ResourceTypeEnum.ASSISTANT])
|
||||
if not ret:
|
||||
return [], 0
|
||||
flow_ids = [one.resource_id for one in ret]
|
||||
|
||||
# 获取用户可见的技能列表
|
||||
if user.is_admin():
|
||||
data, total = FlowDao.get_all_apps(name, status, flow_ids, flow_type, None, None, page, page_size)
|
||||
else:
|
||||
user_role = UserRoleDao.get_user_roles(user.user_id)
|
||||
role_ids = [role.role_id for role in user_role]
|
||||
role_access = RoleAccessDao.get_role_access_batch(role_ids, [AccessType.FLOW, AccessType.WORK_FLOW,
|
||||
AccessType.ASSISTANT_READ])
|
||||
flow_id_extra = []
|
||||
if role_access:
|
||||
flow_id_extra = [access.third_id for access in role_access]
|
||||
data, total = FlowDao.get_all_apps(name, status, flow_ids, flow_type, user.user_id, flow_id_extra, page,
|
||||
page_size)
|
||||
|
||||
# 应用ID列表
|
||||
resource_ids = []
|
||||
# 技能创建用户的ID列表
|
||||
user_ids = []
|
||||
for one in data:
|
||||
one['id'] = one['id']
|
||||
resource_ids.append(one['id'])
|
||||
user_ids.append(one['user_id'])
|
||||
# 获取列表内的用户信息
|
||||
user_infos = UserDao.get_user_by_ids(user_ids)
|
||||
user_dict = {one.user_id: one.user_name for one in user_infos}
|
||||
|
||||
# 获取列表内的版本信息
|
||||
version_infos = FlowVersionDao.get_list_by_flow_ids(resource_ids)
|
||||
flow_versions = {}
|
||||
for one in version_infos:
|
||||
if one.flow_id not in flow_versions:
|
||||
flow_versions[one.flow_id] = []
|
||||
flow_versions[one.flow_id].append(jsonable_encoder(one))
|
||||
|
||||
resource_groups = GroupResourceDao.get_resources_group(None, resource_ids)
|
||||
resource_group_dict = {}
|
||||
for one in resource_groups:
|
||||
if one.third_id not in resource_group_dict:
|
||||
resource_group_dict[one.third_id] = []
|
||||
resource_group_dict[one.third_id].append(one.group_id)
|
||||
|
||||
resource_tag_dict = TagDao.get_tags_by_resource(None, resource_ids)
|
||||
|
||||
# 增加额外的信息
|
||||
for one in data:
|
||||
one['user_name'] = user_dict.get(one['user_id'], one['user_id'])
|
||||
one['write'] = True if user.is_admin() or user.user_id == one['user_id'] else False
|
||||
one['version_list'] = flow_versions.get(one['id'], [])
|
||||
one['group_ids'] = resource_group_dict.get(one['id'], [])
|
||||
one['tags'] = resource_tag_dict.get(one['id'], [])
|
||||
one['logo'] = cls.get_logo_share_link(one['logo'])
|
||||
one['id'] = one['id']
|
||||
|
||||
return data, total
|
||||
|
||||
@classmethod
|
||||
def run_once(cls, login_user: UserPayload, node_input: Dict[str, any], node_data: Dict[any, any]):
|
||||
|
||||
node_data = BaseNodeData(**node_data.get('data', {}))
|
||||
base_callback = BaseCallback()
|
||||
graph_state = GraphState()
|
||||
graph_state.history_memory = ConversationBufferWindowMemory(k=10)
|
||||
node = NodeFactory.instance_node(node_type=node_data.type,
|
||||
node_data=node_data,
|
||||
user_id=login_user.user_id,
|
||||
workflow_id='tmp_workflow_single_node',
|
||||
graph_state=graph_state,
|
||||
target_edges=None,
|
||||
max_steps=233,
|
||||
callback=base_callback)
|
||||
if node_data.type == NodeType.CODE.value:
|
||||
node.handle_input({
|
||||
'code_input': [
|
||||
{
|
||||
'key': k,
|
||||
'value': v,
|
||||
'type': 'input'
|
||||
} for k, v in node_input.items()
|
||||
]
|
||||
})
|
||||
elif node_data.type == NodeType.TOOL.value:
|
||||
user_input = {}
|
||||
for k, v in node_input.items():
|
||||
user_input[k] = v
|
||||
node.handle_input(user_input)
|
||||
else:
|
||||
for key, val in node_input.items():
|
||||
graph_state.set_variable_by_str(key, val)
|
||||
|
||||
exec_id = generate_uuid()
|
||||
result = node._run(exec_id)
|
||||
log_data = node.parse_log(exec_id, result)
|
||||
res = []
|
||||
for one_batch in log_data:
|
||||
ret = []
|
||||
for one in one_batch:
|
||||
if node_data.type == NodeType.QA_RETRIEVER.value and one['key'] != 'retrieved_result':
|
||||
continue
|
||||
if node_data.type == NodeType.RAG.value and one['key'] != 'retrieved_result' and one['type'] != 'variable':
|
||||
continue
|
||||
if node_data.type == NodeType.LLM.value and one['type'] != 'variable':
|
||||
continue
|
||||
if node_data.type == NodeType.AGENT.value and one['type'] not in ['tool', 'variable']:
|
||||
continue
|
||||
if node_data.type == NodeType.CODE.value and one['key'] != 'code_output':
|
||||
continue
|
||||
if node_data.type == NodeType.TOOL.value and one['key'] != 'output':
|
||||
continue
|
||||
ret.append({
|
||||
'key': one['key'],
|
||||
'value': one['value'],
|
||||
'type': one['type']
|
||||
})
|
||||
res.append(ret)
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
def update_flow_status(cls, login_user: UserPayload, flow_id: str, version_id: int, status: int):
|
||||
"""
|
||||
修改工作流状态, 同时修改工作流的当前版本
|
||||
"""
|
||||
db_flow = FlowDao.get_flow_by_id(flow_id)
|
||||
if not db_flow:
|
||||
raise NotFoundError.http_exception()
|
||||
if not login_user.access_check(db_flow.user_id, flow_id, AccessType.WORK_FLOW_WRITE):
|
||||
raise UnAuthorizedError.http_exception()
|
||||
|
||||
version_info = FlowVersionDao.get_version_by_id(version_id)
|
||||
if not version_info or version_info.flow_id != flow_id:
|
||||
raise NotFoundError.http_exception()
|
||||
if status == FlowStatus.ONLINE.value:
|
||||
# workflow的初始化校验
|
||||
try:
|
||||
_ = Workflow(flow_id, login_user.user_id, version_info.data, False,
|
||||
10,
|
||||
10,
|
||||
None)
|
||||
except Exception as e:
|
||||
raise WorkFlowInitError.http_exception(f'workflow init error: {str(e)}')
|
||||
|
||||
FlowVersionDao.change_current_version(flow_id, version_info)
|
||||
db_flow.status = status
|
||||
FlowDao.update_flow(db_flow)
|
||||
return
|
||||
|
||||
@classmethod
|
||||
def convert_chat_response_to_workflow_event(cls, chat_response: ChatResponse) -> WorkflowEvent:
|
||||
workflow_event = WorkflowEvent(
|
||||
event=chat_response.category,
|
||||
message_id=chat_response.message_id,
|
||||
status='end',
|
||||
node_id=chat_response.message.get('node_id'),
|
||||
node_execution_id=chat_response.message.get('unique_id'),
|
||||
)
|
||||
match workflow_event.event:
|
||||
case WorkflowEventType.UserInput.value:
|
||||
return cls.convert_user_input_event(chat_response, workflow_event)
|
||||
case WorkflowEventType.GuideWord.value:
|
||||
workflow_event.output_schema = WorkflowOutputSchema(
|
||||
message=chat_response.message.get('guide_word')
|
||||
)
|
||||
case WorkflowEventType.GuideQuestion.value:
|
||||
workflow_event.output_schema = WorkflowOutputSchema(
|
||||
message=chat_response.message.get('guide_question')
|
||||
)
|
||||
case WorkflowEventType.OutputMsg.value:
|
||||
return cls.convert_output_event(chat_response, workflow_event)
|
||||
case WorkflowEventType.OutputWithChoose.value:
|
||||
return cls.convert_output_choose_event(chat_response, workflow_event)
|
||||
case WorkflowEventType.OutputWithInput.value:
|
||||
return cls.convert_output_input_event(chat_response, workflow_event)
|
||||
case WorkflowEventType.StreamMsg.value:
|
||||
workflow_event.status = chat_response.type
|
||||
workflow_event.output_schema = WorkflowOutputSchema(
|
||||
message=chat_response.message.get('msg'),
|
||||
reasoning_content=chat_response.message.get('reasoning_content'),
|
||||
output_key=chat_response.message.get('output_key'),
|
||||
)
|
||||
cls.handle_source(chat_response, workflow_event)
|
||||
case WorkflowEventType.Error.value:
|
||||
workflow_event.event = WorkflowEventType.Close.value
|
||||
workflow_event.output_schema = WorkflowOutputSchema(
|
||||
message=chat_response.message
|
||||
)
|
||||
|
||||
return workflow_event
|
||||
|
||||
@classmethod
|
||||
def handle_source(cls, chat_response: ChatResponse, workflow_event: WorkflowEvent):
|
||||
if chat_response.source == SourceType.FILE.value:
|
||||
workflow_event.output_schema.source_url = f'resouce/{chat_response.chat_id}/{chat_response.message_id}'
|
||||
elif chat_response.source in [SourceType.LINK.value, SourceType.QA.value]:
|
||||
workflow_event.output_schema.extra = chat_response.extra
|
||||
|
||||
|
||||
@classmethod
|
||||
def convert_user_input_event(cls, chat_response: ChatResponse, workflow_event: WorkflowEvent) -> WorkflowEvent:
|
||||
event_input_schema = chat_response.message.get('input_schema')
|
||||
input_schema = WorkflowInputSchema(
|
||||
input_type=event_input_schema.get('tab'),
|
||||
)
|
||||
if input_schema.input_type == 'form_input':
|
||||
# 前端的表单定义转为后端的表单定义
|
||||
input_schema.value = [WorkflowInputItem(**one) for one in event_input_schema.get('value', [])]
|
||||
for one in input_schema.value:
|
||||
one.label = one.value
|
||||
one.value = ''
|
||||
else:
|
||||
# 说明是输入框输入
|
||||
input_schema.value = [
|
||||
WorkflowInputItem(
|
||||
key=event_input_schema.get('key'),
|
||||
type='text',
|
||||
required=True,
|
||||
value=''
|
||||
)
|
||||
]
|
||||
for one in event_input_schema.get('value', []):
|
||||
tmp = WorkflowInputItem(**one)
|
||||
if tmp.key == 'dialog_files_content':
|
||||
tmp.type = 'dialog_file'
|
||||
tmp.value = []
|
||||
elif tmp.key == 'dialog_file_accept':
|
||||
tmp.type = 'dialog_file_accept'
|
||||
input_schema.value.append(tmp)
|
||||
workflow_event.input_schema = input_schema
|
||||
return workflow_event
|
||||
|
||||
@classmethod
|
||||
def convert_output_event(cls, chat_response: ChatResponse, workflow_event: WorkflowEvent) -> WorkflowEvent:
|
||||
workflow_event.output_schema = WorkflowOutputSchema(
|
||||
message=chat_response.message.get('msg'),
|
||||
files=chat_response.files,
|
||||
output_key=chat_response.message.get('output_key')
|
||||
)
|
||||
cls.handle_source(chat_response, workflow_event)
|
||||
return workflow_event
|
||||
|
||||
@classmethod
|
||||
def convert_output_input_event(cls, chat_response: ChatResponse, workflow_event: WorkflowEvent) -> WorkflowEvent:
|
||||
workflow_event = cls.convert_output_event(chat_response, workflow_event)
|
||||
workflow_event.input_schema = WorkflowInputSchema(
|
||||
input_type='message_inline_input',
|
||||
value=[WorkflowInputItem(
|
||||
key=chat_response.message.get('key'),
|
||||
type='text',
|
||||
required=True,
|
||||
value=chat_response.message.get('input_msg', '')
|
||||
)]
|
||||
)
|
||||
return workflow_event
|
||||
|
||||
@classmethod
|
||||
def convert_output_choose_event(cls, chat_response: ChatResponse, workflow_event: WorkflowEvent) -> WorkflowEvent:
|
||||
workflow_event = cls.convert_output_event(chat_response, workflow_event)
|
||||
workflow_event.input_schema = WorkflowInputSchema(
|
||||
input_type='message_inline_option',
|
||||
value=[WorkflowInputItem(
|
||||
key=chat_response.message.get('key'),
|
||||
type='select',
|
||||
required=True,
|
||||
value='',
|
||||
options=chat_response.message.get('options', [])
|
||||
)]
|
||||
)
|
||||
return workflow_event
|
||||
@@ -0,0 +1,2 @@
|
||||
from .workstation import WorkStationService, WorkstationMessage, WorkstationConversation, SSECallbackClient
|
||||
from .search import SearchTool
|
||||
@@ -0,0 +1,198 @@
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
import requests
|
||||
from bisheng_langchain.gpts.tools.bing_search.tool import BingSearchResults
|
||||
from langchain_community.utilities import BingSearchAPIWrapper
|
||||
|
||||
|
||||
class SearchTool(ABC):
|
||||
"""Abstract base class for search tools."""
|
||||
|
||||
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
self.args = args
|
||||
self.kwargs = kwargs
|
||||
|
||||
def _requests(self, url: str, method: str, **kwargs):
|
||||
"""Base requests method to handle GET and POST requests."""
|
||||
if method == 'GET':
|
||||
response = requests.get(url, **kwargs)
|
||||
elif method == 'POST':
|
||||
response = requests.post(url, **kwargs)
|
||||
else:
|
||||
raise ValueError("Unsupported HTTP method. Use 'GET' or 'POST'.")
|
||||
|
||||
if response.status_code != 200:
|
||||
raise Exception(f"Request {url} failed: {response.status_code} - {response.text}")
|
||||
return response.json()
|
||||
|
||||
# 抽象类
|
||||
@abstractmethod
|
||||
def invoke(self, query: str, **kwargs) -> (str, list):
|
||||
"""
|
||||
Invoke the search tool with the given query.
|
||||
returns
|
||||
- str: The search result as a string.
|
||||
- list: A list of search result link info.
|
||||
"""
|
||||
|
||||
# Here you would implement the actual search logic
|
||||
# For demonstration purposes, we'll just return a dummy response
|
||||
raise NotImplementedError()
|
||||
|
||||
@classmethod
|
||||
def init_search_tool(cls, name: str, *args, **kwargs) -> "SearchTool":
|
||||
"""Initialize the search tool with the given name and arguments."""
|
||||
tool_class: dict = {
|
||||
'bing': BingSearch,
|
||||
'bocha': BoChaSearch,
|
||||
'jina': JinaDeepSearch,
|
||||
'serp': SerpSearch,
|
||||
'tavily': TavilySearch
|
||||
}
|
||||
if name not in tool_class:
|
||||
raise ValueError(f"Tool {name} not found.")
|
||||
c = tool_class[name](*args, **kwargs)
|
||||
return c
|
||||
|
||||
|
||||
class BingSearch(SearchTool):
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.api_key = kwargs.get('api_key')
|
||||
self.base_url = kwargs.get('base_url')
|
||||
|
||||
def invoke(self, query: str, **kwargs) -> (str, list):
|
||||
bingtool = BingSearchResults(api_wrapper=BingSearchAPIWrapper(bing_subscription_key=self.api_key,
|
||||
bing_search_url=self.base_url),
|
||||
num_results=10)
|
||||
res = bingtool.invoke({'query': query})
|
||||
if isinstance(res, str):
|
||||
res = eval(res)
|
||||
search_res = ''
|
||||
web_list = []
|
||||
for index, result in enumerate(res):
|
||||
# 处理搜索结果
|
||||
snippet = result.get('snippet')
|
||||
search_res += f'[webpage ${index} begin]\n${snippet}\n[webpage ${index} end]\n\n'
|
||||
web_list.append({
|
||||
'title': result.get('title'),
|
||||
'url': result.get('link'),
|
||||
'snippet': snippet
|
||||
})
|
||||
return search_res, web_list
|
||||
|
||||
|
||||
class BoChaSearch(SearchTool):
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.api_key = kwargs.get('api_key')
|
||||
self.base_url = 'https://api.bochaai.com/v1/web-search'
|
||||
self.headers = {'Authorization': f'Bearer {self.api_key}'}
|
||||
|
||||
def invoke(self, query: str, **kwargs) -> (str, list):
|
||||
# Implement the search logic for BoCha here
|
||||
# For demonstration purposes, we'll just return a dummy response
|
||||
result = self._requests(self.base_url, method='POST', json={'query': query, 'summary': True}, headers=self.headers)
|
||||
if result.get('code') != 200:
|
||||
raise Exception(f"BoCha Error: {result}")
|
||||
web_pages = result.get('data', {}).get('webPages', {}).get('value', [])
|
||||
# parse result
|
||||
search_res = ''
|
||||
web_list = []
|
||||
for index, item in enumerate(web_pages):
|
||||
search_res += f'[webpage ${index} begin]\n${item.get("snippet")}\n[webpage ${index} end]\n\n'
|
||||
web_list.append({
|
||||
'title': item.get('name'),
|
||||
'url': item.get('url'),
|
||||
'snippet': item.get('snippet')
|
||||
})
|
||||
return search_res, web_list
|
||||
|
||||
|
||||
class JinaDeepSearch(SearchTool):
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.api_key = kwargs.get('api_key')
|
||||
self.base_url = 'https://deepsearch.jina.ai/v1/chat/completions'
|
||||
self.headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.api_key}"
|
||||
}
|
||||
|
||||
def invoke(self, query: str, **kwargs) -> (str, list):
|
||||
req_data={
|
||||
"model": "jina-deepsearch-v1",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": query
|
||||
},
|
||||
],
|
||||
"stream": False,
|
||||
"reasoning_effort": "low",
|
||||
"max_attempts": 1,
|
||||
"no_direct_answer": False
|
||||
}
|
||||
result = self._requests(self.base_url, method='POST', json=req_data, headers=self.headers)
|
||||
|
||||
choices = result.get('choices', [])
|
||||
search_res = ''
|
||||
web_list = []
|
||||
for index, item in enumerate(choices):
|
||||
item_message = item.get('message', {})
|
||||
search_res += f'[webpage ${index} begin]\n${item_message.get("content")}\n[webpage ${index} end]\n\n'
|
||||
for one_web in item_message.get('annotations', []):
|
||||
one_web_info = one_web.get('url_citation', {})
|
||||
web_list.append({
|
||||
'title': one_web_info.get('title'),
|
||||
'url': one_web.get('url'),
|
||||
'snippet': one_web.get('exactQuote')
|
||||
})
|
||||
return search_res, web_list
|
||||
|
||||
class SerpSearch(SearchTool):
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.api_key = kwargs.get('api_key')
|
||||
self.base_url = 'https://serpapi.com/search.json'
|
||||
|
||||
def invoke(self, query: str, **kwargs) -> (str, list):
|
||||
result = self._requests(self.base_url, method='GET', params={'q': query, 'api_key': self.api_key})
|
||||
|
||||
answer_result = result.get('organic_results', [])
|
||||
search_res = ''
|
||||
web_list = []
|
||||
for index, item in enumerate(answer_result):
|
||||
search_res += f'[webpage ${index} begin]\n${item.get("snippet")}\n[webpage ${index} end]\n\n'
|
||||
web_list.append({
|
||||
'title': item.get('title'),
|
||||
'url': item.get('link'),
|
||||
'snippet': item.get('snippet')
|
||||
})
|
||||
return search_res, web_list
|
||||
|
||||
|
||||
class TavilySearch(SearchTool):
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
self.api_key = kwargs.get('api_key')
|
||||
self.base_url = 'https://api.tavily.com/search'
|
||||
self.headers = {'Authorization': f'Bearer {self.api_key}'}
|
||||
|
||||
def invoke(self, query: str, **kwargs) -> (str, list):
|
||||
result = self._requests(self.base_url, method='POST', json={'query': query}, headers=self.headers)
|
||||
|
||||
answers = result.get('results', [])
|
||||
|
||||
# parse result
|
||||
search_res = ''
|
||||
web_list = []
|
||||
for index, item in enumerate(answers):
|
||||
search_res += f'[webpage ${index} begin]\n${item.get("content")}\n[webpage ${index} end]\n\n'
|
||||
web_list.append({
|
||||
'title': item.get('title'),
|
||||
'url': item.get('url'),
|
||||
'snippet': item.get('content')
|
||||
})
|
||||
return search_res, web_list
|
||||
@@ -0,0 +1,270 @@
|
||||
import asyncio
|
||||
import json
|
||||
from datetime import datetime
|
||||
from typing import Optional, Any
|
||||
|
||||
from fastapi import BackgroundTasks, Request
|
||||
from langchain_core.messages import AIMessage, HumanMessage
|
||||
from loguru import logger
|
||||
from openai import BaseModel
|
||||
from pydantic import field_validator
|
||||
|
||||
from bisheng.api.services import knowledge_imp, llm
|
||||
from bisheng.api.services.base import BaseService
|
||||
from bisheng.api.services.knowledge import KnowledgeService
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.v1.schemas import KnowledgeFileOne, KnowledgeFileProcess, WorkstationConfig
|
||||
from bisheng.database.constants import MessageCategory
|
||||
from bisheng.database.models.config import Config, ConfigDao, ConfigKeyEnum
|
||||
from bisheng.database.models.gpts_tools import GptsToolsDao
|
||||
from bisheng.database.models.knowledge import KnowledgeCreate, KnowledgeDao, KnowledgeTypeEnum
|
||||
from bisheng.database.models.message import ChatMessage, ChatMessageDao
|
||||
from bisheng.database.models.session import MessageSession
|
||||
|
||||
|
||||
class WorkStationService(BaseService):
|
||||
|
||||
@classmethod
|
||||
def update_config(cls, request: Request, login_user: UserPayload, data: WorkstationConfig) \
|
||||
-> WorkstationConfig:
|
||||
""" 更新workflow的默认模型配置 """
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.WORKSTATION)
|
||||
if config:
|
||||
config.value = data.model_dump_json()
|
||||
else:
|
||||
config = Config(key=ConfigKeyEnum.WORKSTATION.value, value=json.dumps(data.dict()))
|
||||
ConfigDao.insert_config(config)
|
||||
|
||||
return data
|
||||
|
||||
@classmethod
|
||||
def sync_tool_info(cls, tools: list[dict]) -> list[dict]:
|
||||
""" 同步工具信息 """
|
||||
if not tools:
|
||||
return []
|
||||
tool_type_ids = [t.get("id") for t in tools]
|
||||
tool_type_info = GptsToolsDao.get_all_tool_type(tool_type_ids)
|
||||
exists_tool_type = {t.id: t for t in tool_type_info}
|
||||
tool_info = GptsToolsDao.get_list_by_type(list(exists_tool_type.keys()))
|
||||
exists_tool_info = {t.id: t for t in tool_info}
|
||||
new_tools = []
|
||||
for one in tools:
|
||||
new_one = exists_tool_type.get(one.get("id"))
|
||||
if not new_one:
|
||||
continue
|
||||
one["name"] = new_one.name
|
||||
one["description"] = new_one.description
|
||||
new_children = []
|
||||
for item in one.get("children", []):
|
||||
if not exists_tool_info.get(item.get("id")):
|
||||
continue
|
||||
item["name"] = exists_tool_info[item.get("id")].name
|
||||
item["description"] = exists_tool_info[item.get("id")].desc
|
||||
item["tool_key"] = exists_tool_info[item.get("id")].tool_key
|
||||
new_children.append(item)
|
||||
one["children"] = new_children
|
||||
new_tools.append(one)
|
||||
return new_tools
|
||||
|
||||
@classmethod
|
||||
def parse_config(cls, config: Any) -> Optional[WorkstationConfig]:
|
||||
if config:
|
||||
ret = json.loads(config.value)
|
||||
ret = WorkstationConfig(**ret)
|
||||
if ret.assistantIcon and ret.assistantIcon.relative_path:
|
||||
ret.assistantIcon.image = cls.get_logo_share_link(ret.assistantIcon.relative_path)
|
||||
if ret.sidebarIcon and ret.sidebarIcon.relative_path:
|
||||
ret.sidebarIcon.image = cls.get_logo_share_link(ret.sidebarIcon.relative_path)
|
||||
|
||||
# 兼容旧的websearch配置
|
||||
if ret.webSearch and not ret.webSearch.params:
|
||||
ret.webSearch.tool = 'bing'
|
||||
ret.webSearch.params = {'api_key': ret.webSearch.bingKey, 'base_url': ret.webSearch.bingUrl}
|
||||
# 判断工具是否被删除, 同步工具最新的信息名称和描述等
|
||||
ret.linsightConfig.tools = cls.sync_tool_info(ret.linsightConfig.tools)
|
||||
return ret
|
||||
return None
|
||||
|
||||
@classmethod
|
||||
def get_config(cls) -> WorkstationConfig | None:
|
||||
""" 获取工作台的默认配置 """
|
||||
config = ConfigDao.get_config(ConfigKeyEnum.WORKSTATION)
|
||||
return cls.parse_config(config)
|
||||
|
||||
@classmethod
|
||||
async def aget_config(cls) -> WorkstationConfig | None:
|
||||
""" 异步获取工作台的默认配置 """
|
||||
config = await ConfigDao.aget_config(ConfigKeyEnum.WORKSTATION)
|
||||
return cls.parse_config(config)
|
||||
|
||||
@classmethod
|
||||
async def uploadPersonalKnowledge(
|
||||
cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
file_path,
|
||||
background_tasks: BackgroundTasks,
|
||||
):
|
||||
# 查询是否有个人知识库
|
||||
knowledge = KnowledgeDao.get_user_knowledge(login_user.user_id, None,
|
||||
KnowledgeTypeEnum.PRIVATE)
|
||||
if not knowledge:
|
||||
model = llm.LLMService.get_knowledge_llm()
|
||||
knowledgeCreate = KnowledgeCreate(name='个人知识库',
|
||||
type=KnowledgeTypeEnum.PRIVATE.value,
|
||||
user_id=login_user.user_id,
|
||||
model=model.embedding_model_id)
|
||||
|
||||
knowledge = KnowledgeService.create_knowledge(request, login_user, knowledgeCreate)
|
||||
else:
|
||||
knowledge = knowledge[0]
|
||||
req_data = KnowledgeFileProcess(knowledge_id=knowledge.id,
|
||||
file_list=[KnowledgeFileOne(file_path=file_path)])
|
||||
res = KnowledgeService.process_knowledge_file(request,
|
||||
UserPayload(user_id=login_user.user_id),
|
||||
background_tasks, req_data)
|
||||
return res
|
||||
|
||||
@classmethod
|
||||
def queryKnowledgeList(
|
||||
cls,
|
||||
request: Request,
|
||||
login_user: UserPayload,
|
||||
page: int,
|
||||
size: int,
|
||||
):
|
||||
# 查询是否有个人知识库
|
||||
knowledge = KnowledgeDao.get_user_knowledge(login_user.user_id, None,
|
||||
KnowledgeTypeEnum.PRIVATE)
|
||||
if not knowledge:
|
||||
return [], 0
|
||||
res, total, _ = KnowledgeService.get_knowledge_files(
|
||||
request,
|
||||
UserPayload(user_id=login_user.user_id),
|
||||
knowledge[0].id,
|
||||
page=page,
|
||||
page_size=size)
|
||||
return res, total
|
||||
|
||||
@classmethod
|
||||
def queryChunksFromDB(cls, question: str, login_user: UserPayload):
|
||||
knowledge = KnowledgeDao.get_user_knowledge(login_user.user_id, None,
|
||||
KnowledgeTypeEnum.PRIVATE)
|
||||
|
||||
if not knowledge:
|
||||
return []
|
||||
|
||||
search_kwargs = {'partition_key': knowledge[0].id}
|
||||
embedding = knowledge_imp.decide_embeddings(knowledge[0].model)
|
||||
vectordb = knowledge_imp.decide_vectorstores(knowledge[0].collection_name, 'Milvus',
|
||||
embedding)
|
||||
vectordb.partition_key = knowledge[0].id
|
||||
content = vectordb.as_retriever(search_kwargs=search_kwargs)._get_relevant_documents(
|
||||
question, run_manager=None)
|
||||
if content:
|
||||
content = [
|
||||
knowledge_imp.KnowledgeUtils.chunk2promt(c.page_content, c.metadata)
|
||||
for c in content
|
||||
]
|
||||
else:
|
||||
content = []
|
||||
return content
|
||||
|
||||
@classmethod
|
||||
def get_chat_history(cls, chat_id: str, size: int = 4):
|
||||
chat_history = []
|
||||
messages = ChatMessageDao.get_messages_by_chat_id(chat_id, ['question', 'answer'], size)
|
||||
for one in messages:
|
||||
# bug fix When constructing multi-turn dialogues, the input and response of
|
||||
# the user and the assistant were reversed, leading to incorrect question-and-answer sequences.
|
||||
extra = json.loads(one.extra) or {}
|
||||
content = extra['prompt'] if 'prompt' in extra else one.message
|
||||
if one.category == MessageCategory.QUESTION.value:
|
||||
chat_history.append(HumanMessage(content=content))
|
||||
elif one.category == MessageCategory.ANSWER.value:
|
||||
chat_history.append(AIMessage(content=content))
|
||||
logger.info(f'loaded {len(chat_history)} chat history for chat_id {chat_id}')
|
||||
return chat_history
|
||||
|
||||
|
||||
class WorkstationMessage(BaseModel):
|
||||
messageId: str
|
||||
conversationId: str
|
||||
createdAt: datetime
|
||||
isCreatedByUser: bool
|
||||
model: Optional[str]
|
||||
parentMessageId: Optional[str]
|
||||
sender: str
|
||||
text: str
|
||||
updateAt: datetime
|
||||
files: Optional[list]
|
||||
error: Optional[bool] = False
|
||||
unfinished: Optional[bool] = False
|
||||
|
||||
@field_validator('messageId', mode='before')
|
||||
@classmethod
|
||||
def convert_message_id(cls, value: Any) -> str:
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
return str(value)
|
||||
|
||||
@field_validator('parentMessageId', mode='before')
|
||||
@classmethod
|
||||
def convert_parent_message_id(cls, value: Any) -> str:
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
return str(value)
|
||||
|
||||
@classmethod
|
||||
def from_chat_message(cls, message: ChatMessage):
|
||||
files = json.loads(message.files) if message.files else []
|
||||
return cls(
|
||||
messageId=message.id,
|
||||
conversationId=message.chat_id,
|
||||
createdAt=message.create_time,
|
||||
updateAt=message.update_time,
|
||||
isCreatedByUser=not message.is_bot,
|
||||
model=None,
|
||||
parentMessageId=json.loads(message.extra).get('parentMessageId'),
|
||||
error=json.loads(message.extra).get('error', False),
|
||||
unfinished=json.loads(message.extra).get('unfinished', False),
|
||||
sender=message.sender,
|
||||
text=message.message,
|
||||
files=files,
|
||||
)
|
||||
|
||||
|
||||
class WorkstationConversation(BaseModel):
|
||||
conversationId: str
|
||||
user: str
|
||||
createdAt: datetime
|
||||
updateAt: datetime
|
||||
model: Optional[str]
|
||||
title: Optional[str]
|
||||
|
||||
@classmethod
|
||||
def from_chat_session(cls, session: MessageSession):
|
||||
return cls(
|
||||
conversationId=session.chat_id,
|
||||
user=session.user_id,
|
||||
createdAt=session.create_time,
|
||||
updateAt=session.update_time,
|
||||
model=None,
|
||||
title=session.flow_name,
|
||||
)
|
||||
|
||||
@field_validator('user', mode='before')
|
||||
@classmethod
|
||||
def convert_user(cls, v: Any) -> str:
|
||||
if isinstance(v, str):
|
||||
return v
|
||||
return str(v)
|
||||
|
||||
|
||||
class SSECallbackClient:
|
||||
|
||||
def __init__(self):
|
||||
self.queue = asyncio.Queue()
|
||||
|
||||
async def send_json(self, data):
|
||||
self.queue.put_nowait(data)
|
||||
@@ -1,15 +1,17 @@
|
||||
import hashlib
|
||||
import json
|
||||
import xml.dom.minidom
|
||||
from pathlib import Path
|
||||
from typing import Dict, List
|
||||
|
||||
import aiohttp
|
||||
|
||||
from bisheng.api.v1.schemas import StreamData
|
||||
from bisheng.database.base import session_getter
|
||||
from bisheng.database.models.role_access import AccessType, RoleAccess
|
||||
from bisheng.database.models.variable_value import Variable
|
||||
from bisheng.graph.graph.base import Graph
|
||||
from bisheng.utils.logger import logger
|
||||
from fastapi import Request, WebSocket
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
from platformdirs import user_cache_dir
|
||||
from sqlalchemy import delete
|
||||
from sqlmodel import select
|
||||
@@ -94,7 +96,8 @@ async def build_flow(graph_data: dict,
|
||||
}
|
||||
yield str(StreamData(event='log', data=log_dict))
|
||||
# # 如果存在文件,当前不操作文件,避免重复操作
|
||||
if not process_file and vertex.base_type == 'documentloaders':
|
||||
if not process_file and (vertex.base_type == 'documentloaders'
|
||||
or vertex.base_type == 'input_output'):
|
||||
template_dict = {
|
||||
key: value
|
||||
for key, value in vertex.data['node']['template'].items()
|
||||
@@ -172,7 +175,8 @@ async def build_flow_no_yield(graph_data: dict,
|
||||
for vertex in sorted_vertices:
|
||||
try:
|
||||
# 如果存在文件,当前不操作文件,避免重复操作
|
||||
if not process_file and vertex.base_type == 'documentloaders':
|
||||
if not process_file and (vertex.base_type == 'documentloaders'
|
||||
or vertex.base_type == 'input_output'):
|
||||
template_dict = {
|
||||
key: value
|
||||
for key, value in vertex.data['node']['template'].items()
|
||||
@@ -189,21 +193,26 @@ async def build_flow_no_yield(graph_data: dict,
|
||||
if vertex.base_type == 'vectorstores':
|
||||
# 注入user_name
|
||||
vertex.params['user_name'] = kwargs.get('user_name') if kwargs else ''
|
||||
# 知识库通过参数传参
|
||||
if 'collection_name' in kwargs and 'collection_name' in vertex.params:
|
||||
vertex.params['collection_name'] = kwargs['collection_name']
|
||||
if 'collection_name' in kwargs and 'index_name' in vertex.params:
|
||||
vertex.params['index_name'] = kwargs['collection_name']
|
||||
if vertex.vertex_type not in [
|
||||
'MilvusWithPermissionCheck', 'ElasticsearchWithPermissionCheck'
|
||||
]:
|
||||
# 知识库通过参数传参
|
||||
if 'collection_name' in kwargs and 'collection_name' in vertex.params:
|
||||
vertex.params['collection_name'] = kwargs['collection_name']
|
||||
if 'collection_name' in kwargs and 'index_name' in vertex.params:
|
||||
vertex.params['index_name'] = kwargs['collection_name']
|
||||
|
||||
if 'collection_name' in vertex.params and not vertex.params.get('collection_name'):
|
||||
vertex.params['collection_name'] = f'tmp_{flow_id}_{chat_id if chat_id else 1}'
|
||||
logger.info(f"rename_vector_col col={vertex.params['collection_name']}")
|
||||
if process_file:
|
||||
# L1 清除Milvus历史记录
|
||||
vertex.params['drop_old'] = True
|
||||
elif 'index_name' in vertex.params and not vertex.params.get('index_name'):
|
||||
# es
|
||||
vertex.params['index_name'] = f'tmp_{flow_id}_{chat_id if chat_id else 1}'
|
||||
if 'collection_name' in vertex.params and not vertex.params.get(
|
||||
'collection_name'):
|
||||
vertex.params[
|
||||
'collection_name'] = f'tmp_{flow_id}_{chat_id if chat_id else 1}'
|
||||
logger.info(f"rename_vector_col col={vertex.params['collection_name']}")
|
||||
if process_file:
|
||||
# L1 清除Milvus历史记录
|
||||
vertex.params['drop_old'] = True
|
||||
elif 'index_name' in vertex.params and not vertex.params.get('index_name'):
|
||||
# es
|
||||
vertex.params['index_name'] = f'tmp_{flow_id}_{chat_id if chat_id else 1}'
|
||||
|
||||
if vertex.base_type == 'chains' and 'retriever' in vertex.params:
|
||||
vertex.params['user_name'] = kwargs.get('user_name') if kwargs else ''
|
||||
@@ -222,24 +231,18 @@ async def build_flow_no_yield(graph_data: dict,
|
||||
return graph
|
||||
|
||||
|
||||
def access_check(payload: dict, owner_user_id: int, target_id: int, type: AccessType) -> bool:
|
||||
if payload.get('role') != 'admin':
|
||||
# role_access
|
||||
with session_getter() as session:
|
||||
role_access = session.exec(
|
||||
select(RoleAccess).where(RoleAccess.role_id.in_(payload.get('role')),
|
||||
RoleAccess.type == type.value)).all()
|
||||
third_ids = [access.third_id for access in role_access]
|
||||
if owner_user_id != payload.get('user_id') and str(target_id) not in third_ids:
|
||||
return False
|
||||
return True
|
||||
async def check_permissions(Authorize: AuthJWT, roles: List[str]):
|
||||
Authorize.jwt_required()
|
||||
payload = json.loads(Authorize.get_jwt_subject())
|
||||
user_roles = [payload.get('role')] if isinstance(payload.get('role'),
|
||||
str) else payload.get('role')
|
||||
if any(role in roles for role in user_roles):
|
||||
return True
|
||||
else:
|
||||
raise ValueError('权限不够')
|
||||
|
||||
|
||||
def get_L2_param_from_flow(
|
||||
flow_data: dict,
|
||||
flow_id: str,
|
||||
version_id: int = None
|
||||
):
|
||||
def get_L2_param_from_flow(flow_data: dict, flow_id: str, version_id: int = None):
|
||||
graph = Graph.from_payload(flow_data)
|
||||
node_id = []
|
||||
variable_ids = []
|
||||
@@ -252,8 +255,9 @@ def get_L2_param_from_flow(
|
||||
variable_ids.append(node.id)
|
||||
|
||||
with session_getter() as session:
|
||||
db_variables = session.exec(select(Variable).where(Variable.flow_id == flow_id,
|
||||
Variable.version_id == version_id)).all()
|
||||
db_variables = session.exec(
|
||||
select(Variable).where(Variable.flow_id == flow_id,
|
||||
Variable.version_id == version_id)).all()
|
||||
|
||||
old_file_ids = {
|
||||
variable.node_id: variable
|
||||
@@ -290,9 +294,9 @@ def get_L2_param_from_flow(
|
||||
if update:
|
||||
[session.add(var) for var in update]
|
||||
if delete_node_ids:
|
||||
session.exec(delete(Variable).where(Variable.node_id.in_(delete_node_ids),
|
||||
version_id == version_id,
|
||||
flow_id == flow_id))
|
||||
session.exec(
|
||||
delete(Variable).where(Variable.node_id.in_(delete_node_ids),
|
||||
version_id == version_id, flow_id == flow_id))
|
||||
session.commit()
|
||||
return True
|
||||
except Exception as e:
|
||||
@@ -398,15 +402,15 @@ def parse_gpus(gpu_str: str) -> List[Dict]:
|
||||
'gpu_util')[0]
|
||||
res.append({
|
||||
'gpu_uuid':
|
||||
gpu_uuid_elem.firstChild.data,
|
||||
gpu_uuid_elem.firstChild.data,
|
||||
'gpu_id':
|
||||
gpu_id_elem.firstChild.data,
|
||||
gpu_id_elem.firstChild.data,
|
||||
'gpu_total_mem':
|
||||
'%.2f G' % (float(gpu_total_mem.firstChild.data.split(' ')[0]) / 1024),
|
||||
'%.2f G' % (float(gpu_total_mem.firstChild.data.split(' ')[0]) / 1024),
|
||||
'gpu_used_mem':
|
||||
'%.2f G' % (float(free_mem.firstChild.data.split(' ')[0]) / 1024),
|
||||
'%.2f G' % (float(free_mem.firstChild.data.split(' ')[0]) / 1024),
|
||||
'gpu_utility':
|
||||
round(float(gpu_utility_elem.firstChild.data.split(' ')[0]) / 100, 2)
|
||||
round(float(gpu_utility_elem.firstChild.data.split(' ')[0]) / 100, 2)
|
||||
})
|
||||
return res
|
||||
|
||||
@@ -416,6 +420,20 @@ async def get_url_content(url: str) -> str:
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get(url) as response:
|
||||
if response.status != 200:
|
||||
raise Exception(f"Failed to download content, HTTP status code: {response.status}")
|
||||
raise Exception(f'Failed to download content, HTTP status code: {response.status}')
|
||||
res = await response.read()
|
||||
return res.decode('utf-8')
|
||||
|
||||
|
||||
def get_request_ip(request: Request | WebSocket) -> str:
|
||||
""" 获取客户端真实IP """
|
||||
x_forwarded_for = request.headers.get('X-Forwarded-For')
|
||||
if x_forwarded_for:
|
||||
return x_forwarded_for.split(',')[0]
|
||||
return request.client.host
|
||||
|
||||
|
||||
def md5_hash(original_string: str):
|
||||
md5 = hashlib.md5()
|
||||
md5.update(original_string.encode('utf-8'))
|
||||
return md5.hexdigest()
|
||||
|
||||
@@ -1,17 +1,28 @@
|
||||
from bisheng.api.v1.assistant import router as assistant_router
|
||||
from bisheng.api.v1.audit import router as audit_router
|
||||
from bisheng.api.v1.chat import router as chat_router
|
||||
from bisheng.api.v1.component import router as component_router
|
||||
from bisheng.api.v1.endpoints import router as endpoints_router
|
||||
from bisheng.api.v1.evaluation import router as evaluation_router
|
||||
from bisheng.api.v1.finetune import router as finetune_router
|
||||
from bisheng.api.v1.flows import router as flows_router
|
||||
from bisheng.api.v1.invite_code import router as invite_code_router
|
||||
from bisheng.api.v1.knowledge import router as knowledge_router
|
||||
from bisheng.api.v1.linsight import router as linsight_router
|
||||
from bisheng.api.v1.llm import router as llm_router
|
||||
from bisheng.api.v1.mark_task import router as mark_router
|
||||
from bisheng.api.v1.qa import router as qa_router
|
||||
from bisheng.api.v1.report import router as report_router
|
||||
from bisheng.api.v1.server import router as server_router
|
||||
from bisheng.api.v1.skillcenter import router as skillcenter_router
|
||||
from bisheng.api.v1.tag import router as tag_router
|
||||
from bisheng.api.v1.tool import router as tool_router
|
||||
from bisheng.api.v1.user import router as user_router
|
||||
from bisheng.api.v1.usergroup import router as group_router
|
||||
from bisheng.api.v1.validate import router as validate_router
|
||||
from bisheng.api.v1.variable import router as variable_router
|
||||
from bisheng.api.v1.workflow import router as workflow_router
|
||||
from bisheng.api.v1.workstation import router as workstation_router
|
||||
|
||||
__all__ = [
|
||||
'chat_router',
|
||||
@@ -28,4 +39,15 @@ __all__ = [
|
||||
'finetune_router',
|
||||
'component_router',
|
||||
'assistant_router',
|
||||
'evaluation_router',
|
||||
'group_router',
|
||||
'audit_router',
|
||||
'tag_router',
|
||||
'llm_router',
|
||||
'workflow_router',
|
||||
'mark_router',
|
||||
'workstation_router',
|
||||
"linsight_router",
|
||||
"tool_router",
|
||||
"invite_code_router",
|
||||
]
|
||||
|
||||
@@ -1,97 +1,117 @@
|
||||
import hashlib
|
||||
import json
|
||||
from typing import List, Optional, Any, Dict
|
||||
from uuid import UUID
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import yaml
|
||||
from bisheng_langchain.gpts.tools.api_tools.openapi import OpenApiTools
|
||||
|
||||
from bisheng.api.services.assistant import AssistantService
|
||||
from bisheng.api.services.openapi import OpenApiSchema
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.utils import get_url_content
|
||||
from bisheng.api.v1.schemas import (AssistantCreateReq, AssistantInfo, AssistantUpdateReq,
|
||||
StreamData, UnifiedResponseModel, resp_200, resp_500, DeleteToolTypeReq,
|
||||
TestToolReq)
|
||||
from bisheng.chat.manager import ChatManager
|
||||
from bisheng.chat.types import WorkType
|
||||
from bisheng.database.models.assistant import Assistant
|
||||
from bisheng.database.models.gpts_tools import GptsToolsTypeRead, GptsTools
|
||||
from bisheng.utils.logger import logger
|
||||
from fastapi import APIRouter, Body, Depends, HTTPException, Query, WebSocket, WebSocketException, UploadFile, File
|
||||
from fastapi import (APIRouter, Body, Depends, HTTPException, Query, Request, WebSocket,
|
||||
WebSocketException)
|
||||
from fastapi import status as http_status
|
||||
from fastapi.responses import StreamingResponse
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
|
||||
from bisheng.api.services.assistant import AssistantService
|
||||
from bisheng.api.services.openapi import OpenApiSchema
|
||||
from bisheng.api.services.tool import ToolServices
|
||||
from bisheng.api.services.user_service import UserPayload, get_admin_user, get_login_user
|
||||
from bisheng.api.v1.schemas import (AssistantCreateReq, AssistantUpdateReq,
|
||||
DeleteToolTypeReq, StreamData, TestToolReq,
|
||||
resp_200, resp_500)
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.chat.manager import ChatManager
|
||||
from bisheng.chat.types import WorkType
|
||||
from bisheng.database.constants import ToolPresetType
|
||||
from bisheng.database.models.assistant import Assistant
|
||||
from bisheng.database.models.gpts_tools import GptsToolsTypeRead
|
||||
from bisheng.mcp_manage.manager import ClientManager
|
||||
from bisheng.utils import generate_uuid
|
||||
from bisheng.utils.logger import logger
|
||||
from bisheng_langchain.gpts.tools.api_tools.openapi import OpenApiTools
|
||||
|
||||
router = APIRouter(prefix='/assistant', tags=['Assistant'])
|
||||
chat_manager = ChatManager()
|
||||
|
||||
|
||||
@router.get('', response_model=UnifiedResponseModel[List[AssistantInfo]])
|
||||
@router.get('')
|
||||
def get_assistant(*,
|
||||
name: str = Query(default=None, description='助手名称,模糊匹配, 包含描述的模糊匹配'),
|
||||
tag_id: int = Query(default=None, description='标签ID'),
|
||||
page: Optional[int] = Query(default=1, gt=0, description='页码'),
|
||||
limit: Optional[int] = Query(default=10, gt=0, description='每页条数'),
|
||||
status: Optional[int] = Query(default=None, description='是否上线状态'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return AssistantService.get_assistant(user, name, status, page, limit)
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
return AssistantService.get_assistant(login_user, name, status, tag_id, page, limit)
|
||||
|
||||
|
||||
# 获取某个助手的详细信息
|
||||
@router.get('/info/{assistant_id}', response_model=UnifiedResponseModel[AssistantInfo])
|
||||
def get_assistant_info(*, assistant_id: UUID, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
return AssistantService.get_assistant_info(assistant_id, current_user.get('user_id'))
|
||||
@router.get('/info/{assistant_id}')
|
||||
def get_assistant_info(*, assistant_id: str, login_user: UserPayload = Depends(get_login_user)):
|
||||
"""获取助手信息"""
|
||||
return AssistantService.get_assistant_info(assistant_id, login_user)
|
||||
|
||||
|
||||
@router.post('/delete', response_model=UnifiedResponseModel)
|
||||
def delete_assistant(*, assistant_id: UUID, Authorize: AuthJWT = Depends()):
|
||||
@router.post('/delete')
|
||||
def delete_assistant(*,
|
||||
request: Request,
|
||||
assistant_id: str,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
"""删除助手"""
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return AssistantService.delete_assistant(assistant_id, user)
|
||||
return AssistantService.delete_assistant(request, login_user, assistant_id)
|
||||
|
||||
|
||||
@router.post('', response_model=UnifiedResponseModel[AssistantInfo])
|
||||
async def create_assistant(*, req: AssistantCreateReq, Authorize: AuthJWT = Depends()):
|
||||
@router.post('')
|
||||
async def create_assistant(*,
|
||||
request: Request,
|
||||
req: AssistantCreateReq,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
# get login user
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
|
||||
assistant = Assistant(**req.dict(), user_id=current_user.get('user_id'))
|
||||
return await AssistantService.create_assistant(assistant)
|
||||
assistant = Assistant(**req.dict(), user_id=login_user.user_id)
|
||||
try:
|
||||
return await AssistantService.create_assistant(request, login_user, assistant)
|
||||
except Exception as e:
|
||||
logger.exception('create_assistant error')
|
||||
return resp_500(message=f'创建助手出错:{str(e)}')
|
||||
|
||||
|
||||
@router.put('', response_model=UnifiedResponseModel[AssistantInfo])
|
||||
async def update_assistant(*, req: AssistantUpdateReq, Authorize: AuthJWT = Depends()):
|
||||
@router.put('')
|
||||
async def update_assistant(*,
|
||||
request: Request,
|
||||
req: AssistantUpdateReq,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
# get login user
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return await AssistantService.update_assistant(req, user)
|
||||
return await AssistantService.update_assistant(request, login_user, req)
|
||||
|
||||
|
||||
@router.post('/status', response_model=UnifiedResponseModel)
|
||||
@router.post('/status')
|
||||
async def update_status(*,
|
||||
assistant_id: UUID = Body(description='助手唯一ID', alias='id'),
|
||||
request: Request,
|
||||
assistant_id: str = Body(description='助手唯一ID', alias='id'),
|
||||
status: int = Body(description='是否上线,1:上线,0:下线'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return await AssistantService.update_status(assistant_id, status, user)
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
return await AssistantService.update_status(request, login_user, assistant_id, status)
|
||||
|
||||
|
||||
@router.post('/auto/task')
|
||||
async def auto_update_assistant_task(*, request: Request, login_user: UserPayload = Depends(get_login_user),
|
||||
assistant_id: str = Body(description='助手唯一ID'),
|
||||
prompt: str = Body(description='用户填写的提示词')):
|
||||
# 存入缓存
|
||||
task_id = generate_uuid()
|
||||
redis_client.set(f'auto_update_task:{task_id}', {
|
||||
'assistant_id': assistant_id,
|
||||
'prompt': prompt,
|
||||
})
|
||||
return resp_200(data={
|
||||
'task_id': task_id
|
||||
})
|
||||
|
||||
|
||||
# 自动优化prompt和工具选择
|
||||
@router.get('/auto', response_class=StreamingResponse)
|
||||
async def auto_update_assistant(*,
|
||||
assistant_id: UUID = Query(description='助手唯一ID'),
|
||||
prompt: str = Query(description='用户填写的提示词')):
|
||||
async def auto_update_assistant(*, task_id: str = Query(description='优化任务唯一ID')):
|
||||
task = redis_client.get(f'auto_update_task:{task_id}')
|
||||
if not task:
|
||||
raise HTTPException(status_code=404, detail='task info not found')
|
||||
assistant_id = task['assistant_id']
|
||||
prompt = task['prompt']
|
||||
|
||||
async def event_stream():
|
||||
try:
|
||||
async for message in AssistantService.auto_update_stream(assistant_id, prompt):
|
||||
@@ -109,44 +129,29 @@ async def auto_update_assistant(*,
|
||||
|
||||
|
||||
# 更新助手的提示词
|
||||
@router.post('/prompt', response_model=UnifiedResponseModel)
|
||||
@router.post('/prompt')
|
||||
async def update_prompt(*,
|
||||
assistant_id: UUID = Body(description='助手唯一ID', alias='id'),
|
||||
assistant_id: str = Body(description='助手唯一ID', alias='id'),
|
||||
prompt: str = Body(description='用户使用的prompt'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return AssistantService.update_prompt(assistant_id, prompt, user)
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
return AssistantService.update_prompt(assistant_id, prompt, login_user)
|
||||
|
||||
|
||||
@router.post('/flow', response_model=UnifiedResponseModel)
|
||||
@router.post('/flow')
|
||||
async def update_flow_list(*,
|
||||
assistant_id: UUID = Body(description='助手唯一ID', alias='id'),
|
||||
assistant_id: str = Body(description='助手唯一ID', alias='id'),
|
||||
flow_list: List[str] = Body(description='用户选择的技能列表'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return AssistantService.update_flow_list(assistant_id, flow_list, user)
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
return AssistantService.update_flow_list(assistant_id, flow_list, login_user)
|
||||
|
||||
|
||||
@router.post('/tool', response_model=UnifiedResponseModel)
|
||||
@router.post('/tool')
|
||||
async def update_tool_list(*,
|
||||
assistant_id: UUID = Body(description='助手唯一ID', alias='id'),
|
||||
assistant_id: str = Body(description='助手唯一ID', alias='id'),
|
||||
tool_list: List[int] = Body(description='用户选择的工具列表'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return AssistantService.update_tool_list(assistant_id, tool_list, user)
|
||||
|
||||
|
||||
# 获取助手可用的模型列表
|
||||
@router.get('/models', response_model=UnifiedResponseModel)
|
||||
async def get_models(*, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
return AssistantService.get_models()
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
""" 更新助手选择的工具列表 """
|
||||
return AssistantService.update_tool_list(assistant_id, tool_list, login_user)
|
||||
|
||||
|
||||
# 助手对话的websocket连接
|
||||
@@ -163,12 +168,12 @@ async def chat(*,
|
||||
Authorize._token = t
|
||||
else:
|
||||
Authorize.jwt_required(auth_from='websocket', websocket=websocket)
|
||||
|
||||
payload = Authorize.get_jwt_subject()
|
||||
payload = json.loads(payload)
|
||||
user_id = payload.get('user_id')
|
||||
await chat_manager.dispatch_client(assistant_id, chat_id, user_id, WorkType.GPTS,
|
||||
websocket)
|
||||
login_user = UserPayload(**payload)
|
||||
request = websocket
|
||||
await chat_manager.dispatch_client(request, assistant_id, chat_id, login_user,
|
||||
WorkType.GPTS, websocket)
|
||||
except WebSocketException as exc:
|
||||
logger.error(f'Websocket exception: {str(exc)}')
|
||||
await websocket.close(code=http_status.WS_1011_INTERNAL_ERROR, reason=str(exc))
|
||||
@@ -181,107 +186,105 @@ async def chat(*,
|
||||
await websocket.close(code=http_status.WS_1011_INTERNAL_ERROR, reason=message)
|
||||
|
||||
|
||||
@router.get('/tool_list', response_model=UnifiedResponseModel)
|
||||
def get_tool_list(*, is_preset: Optional[bool] = None, Authorize: AuthJWT = Depends()):
|
||||
@router.get('/tool_list')
|
||||
def get_tool_list(*,
|
||||
is_preset: Optional[int | bool] = None,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
"""查询所有可见的tool 列表"""
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
return resp_200(AssistantService.get_gpts_tools(current_user.get('user_id'), is_preset))
|
||||
if is_preset is not None and type(is_preset) == bool:
|
||||
is_preset = ToolPresetType.PRESET.value if is_preset else ToolPresetType.API.value
|
||||
return resp_200(AssistantService.get_gpts_tools(login_user, is_preset))
|
||||
|
||||
|
||||
@router.post('/tool_schema', response_model=UnifiedResponseModel)
|
||||
async def get_tool_schema(*,
|
||||
@router.post('/tool/config')
|
||||
async def update_tool_config(*,
|
||||
login_user: UserPayload = Depends(get_admin_user),
|
||||
tool_id: int = Body(description='工具类别唯一ID'),
|
||||
extra: dict = Body(description='工具配置项')):
|
||||
""" 更新工具的配置 """
|
||||
data = AssistantService.update_tool_config(login_user, tool_id, extra)
|
||||
return resp_200(data=data)
|
||||
|
||||
|
||||
@router.post('/tool_schema')
|
||||
async def get_tool_schema(request: Request, login_user: UserPayload = Depends(get_login_user),
|
||||
download_url: Optional[str] = Body(default=None,
|
||||
description='下载url不为空的话优先用下载url'),
|
||||
file_content: Optional[str] = Body(default=None, description='上传的文件'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
file_content: Optional[str] = Body(default=None, description='上传的文件')):
|
||||
""" 下载或者解析openapi schema的内容 转为助手自定义工具的格式 """
|
||||
if download_url:
|
||||
try:
|
||||
file_content = await get_url_content(download_url)
|
||||
except Exception as e:
|
||||
logger.exception(f'file {download_url} download error')
|
||||
return resp_500(message="url文件下载失败:" + str(e))
|
||||
services = ToolServices(request=request, login_user=login_user)
|
||||
tool_type = await services.parse_openapi_schema(download_url, file_content)
|
||||
return resp_200(data=tool_type)
|
||||
|
||||
if not file_content:
|
||||
return resp_500(message="schema内容不能为空")
|
||||
# 根据文件内容是否以`{`开头判断用什么解析方式
|
||||
|
||||
@router.post('/mcp/tool_schema')
|
||||
async def get_mcp_tool_schema(request: Request, login_user: UserPayload = Depends(get_login_user),
|
||||
file_content: Optional[str] = Body(default=None, embed=True,
|
||||
description='mcp服务配置内容')):
|
||||
""" 解析mcp的工具配置文件 """
|
||||
services = ToolServices(request=request, login_user=login_user)
|
||||
tool_type = await services.parse_mcp_schema(file_content)
|
||||
return resp_200(data=tool_type)
|
||||
|
||||
|
||||
@router.post('/mcp/tool_test')
|
||||
async def mcp_tool_run(login_user: UserPayload = Depends(get_login_user),
|
||||
req: TestToolReq = None):
|
||||
""" 测试mcp服务的工具 """
|
||||
try:
|
||||
if file_content.startswith("{"):
|
||||
res = json.loads(file_content)
|
||||
else:
|
||||
res = yaml.safe_load(file_content)
|
||||
# 实例化mcp服务对象,获取工具列表
|
||||
client = await ClientManager.connect_mcp_from_json(req.openapi_schema)
|
||||
extra = json.loads(req.extra)
|
||||
tool_name = extra.get('name')
|
||||
resp = await client.call_tool(tool_name, req.request_params)
|
||||
return resp_200(data=resp)
|
||||
except Exception as e:
|
||||
logger.exception(f'openapi schema parse error')
|
||||
return resp_500(message=f"openapi schema解析报错,请检查内容是否符合json或者yaml格式: {str(e)}")
|
||||
|
||||
# 解析openapi schema转为助手工具的格式
|
||||
try:
|
||||
schema = OpenApiSchema(res)
|
||||
schema.parse_server()
|
||||
if not schema.default_server.startswith(("http", "https")):
|
||||
return resp_500(message=f"server中的url必须以http或者https开头: {schema.default_server}")
|
||||
tool_type = GptsToolsTypeRead(name=schema.title, description=schema.description,
|
||||
is_preset=0, is_delete=0, server_host=schema.default_server,
|
||||
openapi_schema=file_content, children=[])
|
||||
# 解析获取所有的api
|
||||
schema.parse_paths()
|
||||
for one in schema.apis:
|
||||
tool_type.children.append(GptsTools(
|
||||
name=one['operationId'],
|
||||
desc=one['description'],
|
||||
tool_key=hashlib.md5(one['operationId'].encode("utf-8")).hexdigest(),
|
||||
is_preset=0,
|
||||
is_delete=0,
|
||||
api_params=one["parameters"],
|
||||
extra=json.dumps(one, ensure_ascii=False),
|
||||
))
|
||||
return resp_200(data=tool_type)
|
||||
except Exception as e:
|
||||
logger.exception(f'openapi schema parse error')
|
||||
return resp_500(message="openapi schema解析失败:" + str(e))
|
||||
logger.exception('mcp_tool_run error')
|
||||
return resp_500(message=f'测试请求出错:{str(e)}')
|
||||
|
||||
|
||||
@router.post('/tool_list', response_model=UnifiedResponseModel[GptsToolsTypeRead])
|
||||
def add_tool_type(*, req: Dict = Body(default={}, description="openapi解析后的工具对象"),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
@router.post('/mcp/refresh')
|
||||
async def refresh_all_mcp_tools(request: Request, login_user: UserPayload = Depends(get_login_user)):
|
||||
""" 刷新用户当前所有的mcp工具列表 """
|
||||
services = ToolServices(request=request, login_user=login_user)
|
||||
error_msg = await services.refresh_all_mcp()
|
||||
if error_msg:
|
||||
return resp_500(message=error_msg)
|
||||
return resp_200(message='刷新成功')
|
||||
|
||||
|
||||
@router.post('/tool_list')
|
||||
async def add_tool_type(*,
|
||||
req: Dict = Body(default={}, description='openapi解析后的工具对象'),
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
""" 新增自定义tool """
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
req = GptsToolsTypeRead(**req)
|
||||
return AssistantService.add_gpts_tools(user, req)
|
||||
return await AssistantService.add_gpts_tools(login_user, req)
|
||||
|
||||
|
||||
@router.put('/tool_list', response_model=UnifiedResponseModel[GptsToolsTypeRead])
|
||||
def update_tool_type(*, req: Dict = Body(default={}, description="通过openapi 解析后的内容,包含类别的唯一ID"),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
@router.put('/tool_list')
|
||||
async def update_tool_type(*,
|
||||
login_user: UserPayload = Depends(get_login_user),
|
||||
req: Dict = Body(default={}, description='通过openapi 解析后的内容,包含类别的唯一ID')):
|
||||
""" 更新自定义tool """
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
req = GptsToolsTypeRead(**req)
|
||||
return AssistantService.update_gpts_tools(user, req)
|
||||
return resp_200(data=await ToolServices.update_gpts_tools(login_user, req))
|
||||
|
||||
|
||||
@router.delete('/tool_list', response_model=UnifiedResponseModel)
|
||||
def delete_tool_type(*, req: DeleteToolTypeReq, Authorize: AuthJWT = Depends()):
|
||||
@router.delete('/tool_list')
|
||||
def delete_tool_type(*, login_user: UserPayload = Depends(get_login_user), req: DeleteToolTypeReq):
|
||||
""" 删除自定义工具 """
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
return AssistantService.delete_gpts_tools(user, req.tool_type_id)
|
||||
return AssistantService.delete_gpts_tools(login_user, req.tool_type_id)
|
||||
|
||||
|
||||
@router.post('/tool_test', response_model=UnifiedResponseModel)
|
||||
async def test_tool_type(*, req: TestToolReq, Authorize: AuthJWT = Depends()):
|
||||
@router.post('/tool_test')
|
||||
async def tool_run(*, login_user: UserPayload = Depends(get_login_user), req: TestToolReq):
|
||||
""" 测试自定义工具 """
|
||||
Authorize.jwt_required()
|
||||
current_user = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**current_user)
|
||||
|
||||
tool_params = OpenApiSchema.parse_openapi_tool_params('test', 'test', req.extra, req.server_host,
|
||||
req.auth_method, req.auth_type, req.api_key)
|
||||
extra = json.loads(req.extra)
|
||||
extra.update({'api_location': req.api_location, 'parameter_name': req.parameter_name})
|
||||
tool_params = OpenApiSchema.parse_openapi_tool_params('test', 'test', json.dumps(extra),
|
||||
req.server_host, req.auth_method,
|
||||
req.auth_type, req.api_key)
|
||||
|
||||
openapi_tool = OpenApiTools.get_api_tool('test', **tool_params)
|
||||
try:
|
||||
@@ -289,4 +292,4 @@ async def test_tool_type(*, req: TestToolReq, Authorize: AuthJWT = Depends()):
|
||||
return resp_200(data=resp)
|
||||
except Exception as e:
|
||||
logger.exception('tool_test error')
|
||||
return resp_500(message=f"测试请求出错:{str(e)}")
|
||||
return resp_500(message=f'测试请求出错:{str(e)}')
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
from datetime import datetime
|
||||
from typing import Optional, List
|
||||
|
||||
from fastapi import APIRouter, Query, Depends
|
||||
|
||||
from bisheng.api.services.audit_log import AuditLogService
|
||||
from bisheng.api.services.user_service import UserPayload, get_login_user
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel, resp_200
|
||||
|
||||
router = APIRouter(prefix='/audit', tags=['AuditLog'])
|
||||
|
||||
|
||||
@router.get('')
|
||||
def get_audit_logs(*,
|
||||
group_ids: Optional[List[str]] = Query(default=[], description='分组id列表'),
|
||||
operator_ids: Optional[List[int]] = Query(default=[], description='操作人id列表'),
|
||||
start_time: Optional[datetime] = Query(default=None, description='开始时间'),
|
||||
end_time: Optional[datetime] = Query(default=None, description='结束时间'),
|
||||
system_id: Optional[str] = Query(default=None, description='系统模块'),
|
||||
event_type: Optional[str] = Query(default=None, description='操作行为'),
|
||||
page: Optional[int] = Query(default=0, description='页码'),
|
||||
limit: Optional[int] = Query(default=0, description='每页条数'),
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
group_ids = [one for one in group_ids if one]
|
||||
operator_ids = [one for one in operator_ids if one]
|
||||
return AuditLogService.get_audit_log(login_user, group_ids, operator_ids,
|
||||
start_time, end_time, system_id, event_type, page, limit)
|
||||
|
||||
|
||||
@router.get('/operators')
|
||||
def get_all_operators(*, login_user: UserPayload = Depends(get_login_user)):
|
||||
"""
|
||||
获取操作过组下资源的所有用户
|
||||
"""
|
||||
return AuditLogService.get_all_operators(login_user)
|
||||
|
||||
|
||||
@router.get('/session')
|
||||
def get_session_list(login_user: UserPayload = Depends(get_login_user),
|
||||
flow_ids: Optional[List[str]] = Query(default=[], description='应用id列表'),
|
||||
user_ids: Optional[List[int]] = Query(default=[], description='用户id列表'),
|
||||
group_ids: Optional[List[int]] = Query(default=[], description='用户组id列表'),
|
||||
start_date: Optional[datetime] = Query(default=None, description='开始时间'),
|
||||
end_date: Optional[datetime] = Query(default=None, description='结束时间'),
|
||||
feedback: Optional[str] = Query(default=None, description='like:点赞;dislike:点踩;copied:复制'),
|
||||
sensitive_status: Optional[int] = Query(default=None, description='敏感词审查状态'),
|
||||
page: Optional[int] = Query(default=1, description='页码'),
|
||||
page_size: Optional[int] = Query(default=10, description='每页条数')):
|
||||
""" 筛选所有会话列表 """
|
||||
data, total = AuditLogService.get_session_list(login_user, flow_ids, user_ids, group_ids, start_date, end_date,
|
||||
feedback, sensitive_status, page, page_size)
|
||||
return resp_200(data={
|
||||
'data': data,
|
||||
'total': total
|
||||
})
|
||||
|
||||
|
||||
@router.get('/session/export')
|
||||
def export_session_messages(login_user: UserPayload = Depends(get_login_user),
|
||||
flow_ids: Optional[List[str]] = Query(default=[], description='应用id列表'),
|
||||
user_ids: Optional[List[int]] = Query(default=[], description='用户id列表'),
|
||||
group_ids: Optional[List[int]] = Query(default=[], description='用户组id列表'),
|
||||
start_date: Optional[datetime] = Query(default=None, description='开始时间'),
|
||||
end_date: Optional[datetime] = Query(default=None, description='结束时间'),
|
||||
feedback: Optional[str] = Query(default=None,
|
||||
description='like:点赞;dislike:点踩;copied:复制'),
|
||||
sensitive_status: Optional[int] = Query(default=None, description='敏感词审查状态')):
|
||||
""" 导出会话详情列表的csv文件 """
|
||||
url = AuditLogService.export_session_messages(login_user, flow_ids, user_ids, group_ids, start_date, end_date,
|
||||
feedback, sensitive_status)
|
||||
return resp_200(data={
|
||||
'url': url
|
||||
})
|
||||
|
||||
|
||||
@router.get('/session/export/data')
|
||||
def get_session_messages(login_user: UserPayload = Depends(get_login_user),
|
||||
flow_ids: Optional[List[str]] = Query(default=[], description='应用id列表'),
|
||||
user_ids: Optional[List[int]] = Query(default=[], description='用户id列表'),
|
||||
group_ids: Optional[List[int]] = Query(default=[], description='用户组id列表'),
|
||||
start_date: Optional[datetime] = Query(default=None, description='开始时间'),
|
||||
end_date: Optional[datetime] = Query(default=None, description='结束时间'),
|
||||
feedback: Optional[str] = Query(default=None,
|
||||
description='like:点赞;dislike:点踩;copied:复制'),
|
||||
sensitive_status: Optional[int] = Query(default=None, description='敏感词审查状态')):
|
||||
""" 导出会话详情列表的数据 """
|
||||
result = AuditLogService.get_session_messages(login_user, flow_ids, user_ids, group_ids, start_date, end_date,
|
||||
feedback, sensitive_status)
|
||||
return resp_200(data={
|
||||
'data': result
|
||||
})
|
||||
@@ -1,7 +1,7 @@
|
||||
from bisheng.interface.utils import extract_input_variables_from_prompt
|
||||
from bisheng.template.frontend_node.base import FrontendNode
|
||||
from langchain.prompts import PromptTemplate
|
||||
from pydantic import BaseModel, validator
|
||||
from pydantic import field_validator, BaseModel
|
||||
|
||||
|
||||
class CacheResponse(BaseModel):
|
||||
@@ -27,11 +27,13 @@ class CodeValidationResponse(BaseModel):
|
||||
imports: dict
|
||||
function: dict
|
||||
|
||||
@validator('imports')
|
||||
@field_validator('imports')
|
||||
@classmethod
|
||||
def validate_imports(cls, v):
|
||||
return v or {'errors': []}
|
||||
|
||||
@validator('function')
|
||||
@field_validator('function')
|
||||
@classmethod
|
||||
def validate_function(cls, v):
|
||||
return v or {'errors': []}
|
||||
|
||||
|
||||
@@ -1,24 +1,33 @@
|
||||
import asyncio
|
||||
import copy
|
||||
import json
|
||||
from queue import Queue
|
||||
from typing import Any, Dict, List, Union
|
||||
|
||||
from bisheng.api.v1.schemas import ChatResponse
|
||||
from bisheng.database.models.message import ChatMessage as ChatMessageModel
|
||||
from bisheng.database.models.message import ChatMessageDao
|
||||
from bisheng.utils.logger import logger
|
||||
from fastapi import WebSocket
|
||||
from langchain.callbacks.base import AsyncCallbackHandler, BaseCallbackHandler
|
||||
from langchain.schema import AgentFinish, LLMResult
|
||||
from langchain.schema.agent import AgentAction
|
||||
from langchain.schema.document import Document
|
||||
from langchain.schema.messages import BaseMessage
|
||||
from langchain_core.messages import ToolMessage
|
||||
|
||||
from bisheng.api.v1.schemas import ChatResponse
|
||||
from bisheng.database.models.message import ChatMessage as ChatMessageModel
|
||||
from bisheng.database.models.message import ChatMessageDao
|
||||
from bisheng.utils.logger import logger
|
||||
|
||||
|
||||
# https://github.com/hwchase17/chat-langchain/blob/master/callback.py
|
||||
class AsyncStreamingLLMCallbackHandler(AsyncCallbackHandler):
|
||||
"""Callback handler for streaming LLM responses."""
|
||||
|
||||
def __init__(self, websocket: WebSocket, flow_id: str, chat_id: str, user_id: int = None):
|
||||
def __init__(self,
|
||||
websocket: WebSocket,
|
||||
flow_id: str,
|
||||
chat_id: str,
|
||||
user_id: int = None,
|
||||
**kwargs: Any):
|
||||
self.websocket = websocket
|
||||
self.flow_id = flow_id
|
||||
self.chat_id = chat_id
|
||||
@@ -33,13 +42,32 @@ class AsyncStreamingLLMCallbackHandler(AsyncCallbackHandler):
|
||||
# }, # 存储工具调用的input信息
|
||||
# }
|
||||
|
||||
# 流式输出的队列
|
||||
self.stream_queue: Queue = kwargs.get('stream_queue')
|
||||
|
||||
async def on_llm_new_token(self, token: str, **kwargs: Any) -> None:
|
||||
logger.debug(f'on_llm_new_token token={token} kwargs={kwargs}')
|
||||
resp = ChatResponse(message=token,
|
||||
type='stream',
|
||||
flow_id=self.flow_id,
|
||||
chat_id=self.chat_id)
|
||||
chunk = kwargs.get('chunk')
|
||||
# azure偶尔会返回一个None
|
||||
if token is None and chunk is None:
|
||||
return
|
||||
reasoning_content = getattr(chunk.message, 'additional_kwargs',
|
||||
{}).get('reasoning_content')
|
||||
if token is None:
|
||||
token = ''
|
||||
resp = ChatResponse(message={
|
||||
'content': token,
|
||||
'reasoning_content': reasoning_content
|
||||
},
|
||||
type='stream',
|
||||
flow_id=self.flow_id,
|
||||
chat_id=self.chat_id)
|
||||
# 将流式输出内容放入到队列内,以方便中断流式输出后,可以将内容记录到数据库
|
||||
await self.websocket.send_json(resp.dict())
|
||||
if self.stream_queue:
|
||||
if reasoning_content:
|
||||
self.stream_queue.put({'type': 'reasoning', 'content': reasoning_content})
|
||||
if token:
|
||||
self.stream_queue.put({'type': 'answer', 'content': token})
|
||||
|
||||
async def on_llm_start(self, serialized: Dict[str, Any], prompts: List[str],
|
||||
**kwargs: Any) -> Any:
|
||||
@@ -63,8 +91,10 @@ class AsyncStreamingLLMCallbackHandler(AsyncCallbackHandler):
|
||||
async def on_chain_end(self, outputs: Dict[str, Any], **kwargs: Any) -> Any:
|
||||
"""Run when chain ends running."""
|
||||
logger.debug(f'on_chain_end outputs={outputs} kwargs={kwargs}')
|
||||
outputs.pop('source_documents', '')
|
||||
logger.info('k=s act=on_chain_end flow_id={} output_dict={}', self.flow_id, outputs)
|
||||
tmp_output = copy.deepcopy(outputs)
|
||||
if isinstance(tmp_output, dict):
|
||||
tmp_output.pop('source_documents', '')
|
||||
logger.info('k=s act=on_chain_end flow_id={} output_dict={}', self.flow_id, tmp_output)
|
||||
|
||||
async def on_chain_error(self, error: Union[Exception, KeyboardInterrupt],
|
||||
**kwargs: Any) -> Any:
|
||||
@@ -92,10 +122,10 @@ class AsyncStreamingLLMCallbackHandler(AsyncCallbackHandler):
|
||||
observation_prefix = kwargs.get('observation_prefix', 'Tool output: ')
|
||||
# from langchain.docstore.document import Document # noqa
|
||||
# result = eval(output).get('result')
|
||||
result = output
|
||||
result = output if isinstance(output, str) else getattr(output, 'content', output)
|
||||
|
||||
# Create a formatted message.
|
||||
intermediate_steps = f'{observation_prefix}{result}'
|
||||
intermediate_steps = f'{observation_prefix}{result[:100]}'
|
||||
|
||||
# Create a ChatResponse instance.
|
||||
resp = ChatResponse(type='stream',
|
||||
@@ -210,9 +240,10 @@ class AsyncStreamingLLMCallbackHandler(AsyncCallbackHandler):
|
||||
# todo 判断技能权限
|
||||
logger.debug(f'on_retriever_end result={result} kwargs={kwargs}')
|
||||
if result:
|
||||
[doc.metadata.pop('bbox', '') for doc in result]
|
||||
tmp_result = copy.deepcopy(result)
|
||||
[doc.metadata.pop('bbox', '') for doc in tmp_result]
|
||||
logger.info('k=s act=on_retriever_end flow_id={} result_without_bbox={}', self.flow_id,
|
||||
result)
|
||||
tmp_result)
|
||||
|
||||
async def on_chat_model_start(self, serialized: Dict[str, Any],
|
||||
messages: List[List[BaseMessage]], **kwargs: Any) -> Any:
|
||||
@@ -228,12 +259,23 @@ class AsyncStreamingLLMCallbackHandler(AsyncCallbackHandler):
|
||||
class StreamingLLMCallbackHandler(BaseCallbackHandler):
|
||||
"""Callback handler for streaming LLM responses."""
|
||||
|
||||
def __init__(self, websocket: WebSocket, flow_id: str, chat_id: str):
|
||||
def __init__(self,
|
||||
websocket: WebSocket,
|
||||
flow_id: str,
|
||||
chat_id: str,
|
||||
user_id: int = None,
|
||||
**kwargs: Any):
|
||||
self.websocket = websocket
|
||||
self.flow_id = flow_id
|
||||
self.chat_id = chat_id
|
||||
self.user_id = user_id
|
||||
|
||||
self.stream_queue: Queue = kwargs.get('stream_queue')
|
||||
|
||||
def on_llm_new_token(self, token: str, **kwargs: Any) -> None:
|
||||
# azure偶尔会返回一个None
|
||||
if token is None:
|
||||
return
|
||||
resp = ChatResponse(message=token,
|
||||
type='stream',
|
||||
flow_id=self.flow_id,
|
||||
@@ -243,6 +285,9 @@ class StreamingLLMCallbackHandler(BaseCallbackHandler):
|
||||
coroutine = self.websocket.send_json(resp.dict())
|
||||
asyncio.run_coroutine_threadsafe(coroutine, loop)
|
||||
|
||||
if self.stream_queue:
|
||||
self.stream_queue.put(token)
|
||||
|
||||
def on_agent_action(self, action: AgentAction, **kwargs: Any) -> Any:
|
||||
log = f'\nThought: {action.log}'
|
||||
# if there are line breaks, split them and send them
|
||||
@@ -289,7 +334,7 @@ class StreamingLLMCallbackHandler(BaseCallbackHandler):
|
||||
|
||||
# from langchain.docstore.document import Document # noqa
|
||||
# result = eval(output).get('result')
|
||||
result = output
|
||||
result = output if isinstance(output, str) else getattr(output, 'content', output)
|
||||
# Create a formatted message.
|
||||
intermediate_steps = f'{observation_prefix}{result}'
|
||||
|
||||
@@ -319,9 +364,10 @@ class StreamingLLMCallbackHandler(BaseCallbackHandler):
|
||||
# todo 判断技能权限
|
||||
logger.debug(f'retriver_result result={result}')
|
||||
if result:
|
||||
[doc.metadata.pop('bbox', '') for doc in result]
|
||||
tmp_result = copy.deepcopy(result)
|
||||
[doc.metadata.pop('bbox', '') for doc in tmp_result]
|
||||
logger.info('k=s act=on_retriever_end flow_id={} result_without_bbox={}', self.flow_id,
|
||||
result)
|
||||
tmp_result)
|
||||
|
||||
def on_chain_start(self, serialized: Dict[str, Any], inputs: Dict[str, Any],
|
||||
**kwargs: Any) -> Any:
|
||||
@@ -332,8 +378,10 @@ class StreamingLLMCallbackHandler(BaseCallbackHandler):
|
||||
def on_chain_end(self, outputs: Dict[str, Any], **kwargs: Any) -> Any:
|
||||
"""Run when chain ends running."""
|
||||
logger.debug(f'on_chain_end outputs={outputs}')
|
||||
outputs.pop('source_documents', '')
|
||||
logger.info('k=s act=on_chain_end flow_id={} output_dict={}', self.flow_id, outputs)
|
||||
tmp_output = copy.deepcopy(outputs)
|
||||
if isinstance(tmp_output, dict):
|
||||
tmp_output.pop('source_documents', '')
|
||||
logger.info('k=s act=on_chain_end flow_id={} output_dict={}', self.flow_id, tmp_output)
|
||||
|
||||
def on_chat_model_start(self, serialized: Dict[str, Any], messages: List[List[BaseMessage]],
|
||||
**kwargs: Any) -> Any:
|
||||
@@ -449,18 +497,18 @@ class AsyncGptsDebugCallbackHandler(AsyncGptsLLMCallbackHandler):
|
||||
extra=json.dumps({'run_id': kwargs.get('run_id').hex}))
|
||||
await self.websocket.send_json(resp.dict())
|
||||
|
||||
async def on_tool_end(self, output: str, **kwargs: Any) -> Any:
|
||||
async def on_tool_end(self, output: ToolMessage, **kwargs: Any) -> Any:
|
||||
"""Run when tool ends running."""
|
||||
logger.debug(f'on_tool_end output={output} kwargs={kwargs}')
|
||||
observation_prefix = kwargs.get('observation_prefix', 'Tool output: ')
|
||||
|
||||
result = output
|
||||
result = output if isinstance(output, str) else getattr(output, 'content', output)
|
||||
# Create a formatted message.
|
||||
intermediate_steps = f'{observation_prefix}\n\n{result}'
|
||||
tool_name, tool_category = self.parse_tool_category(kwargs.get('name'))
|
||||
|
||||
# Create a ChatResponse instance.
|
||||
output_info = {'tool_key': tool_name, 'output': output}
|
||||
output_info = {'tool_key': tool_name, 'output': result}
|
||||
resp = ChatResponse(type='end',
|
||||
category=tool_category,
|
||||
intermediate_steps=intermediate_steps,
|
||||
@@ -473,6 +521,10 @@ class AsyncGptsDebugCallbackHandler(AsyncGptsLLMCallbackHandler):
|
||||
# 从tool cache中获取input信息
|
||||
input_info = self.tool_cache.get(kwargs.get('run_id').hex)
|
||||
if input_info:
|
||||
if not self.chat_id:
|
||||
# 说明是调试界面,不用持久化数据
|
||||
self.tool_cache.pop(kwargs.get('run_id').hex)
|
||||
return
|
||||
output_info.update(input_info['input'])
|
||||
intermediate_steps = f'{input_info["steps"]}\n\n{intermediate_steps}'
|
||||
ChatMessageDao.insert_one(
|
||||
@@ -505,6 +557,10 @@ class AsyncGptsDebugCallbackHandler(AsyncGptsLLMCallbackHandler):
|
||||
await self.websocket.send_json(resp.dict())
|
||||
|
||||
# 保存工具调用记录
|
||||
if not self.chat_id:
|
||||
# 说明是调试界面,不用持久化数据
|
||||
self.tool_cache.pop(kwargs.get('run_id').hex)
|
||||
return
|
||||
tool_name, tool_category = self.parse_tool_category(kwargs.get('name'))
|
||||
self.tool_cache.pop(kwargs.get('run_id').hex)
|
||||
ChatMessageDao.insert_one(
|
||||
@@ -512,7 +568,7 @@ class AsyncGptsDebugCallbackHandler(AsyncGptsLLMCallbackHandler):
|
||||
is_bot=1,
|
||||
message=json.dumps(output_info),
|
||||
intermediate_steps=f'{input_info["steps"]}\n\nTool output:\n\n Error: ' +
|
||||
str(error),
|
||||
str(error),
|
||||
category=tool_category,
|
||||
type='end',
|
||||
flow_id=self.flow_id,
|
||||
|
||||
+434
-139
@@ -1,31 +1,45 @@
|
||||
import json
|
||||
from typing import List, Optional
|
||||
from uuid import UUID
|
||||
from uuid import UUID, uuid4
|
||||
|
||||
from bisheng.api.services.assistant import AssistantService
|
||||
from fastapi import (APIRouter, Body, HTTPException, Query, Request, WebSocket, WebSocketException,
|
||||
status)
|
||||
from fastapi.params import Depends
|
||||
from fastapi.responses import StreamingResponse
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
from sqlmodel import select
|
||||
|
||||
from bisheng.api.errcode.base import NotFoundError
|
||||
from bisheng.api.services import chat_imp
|
||||
from bisheng.api.services.audit_log import AuditLogService
|
||||
from bisheng.api.services.base import BaseService
|
||||
from bisheng.api.services.chat_imp import comment_answer
|
||||
from bisheng.api.services.knowledge_imp import delete_es, delete_vector
|
||||
from bisheng.api.services.user_service import UserPayload
|
||||
from bisheng.api.utils import build_flow, build_input_keys_response
|
||||
from bisheng.api.v1.schemas import (BuildStatus, BuiltResponse, ChatInput, ChatList,
|
||||
FlowGptsOnlineList, InitResponse, StreamData,
|
||||
from bisheng.api.services.user_service import UserPayload, get_login_user
|
||||
from bisheng.api.services.workflow import WorkFlowService
|
||||
from bisheng.api.utils import build_flow, build_input_keys_response, get_request_ip
|
||||
from bisheng.api.v1.schema.base_schema import PageList
|
||||
from bisheng.api.v1.schema.chat_schema import APIChatCompletion, AppChatList
|
||||
from bisheng.api.v1.schema.workflow import WorkflowEventType
|
||||
from bisheng.api.v1.schemas import (AddChatMessages, BuildStatus, BuiltResponse, ChatInput,
|
||||
ChatList, InitResponse, StreamData,
|
||||
UnifiedResponseModel, resp_200)
|
||||
from bisheng.cache.redis import redis_client
|
||||
from bisheng.chat.manager import ChatManager
|
||||
from bisheng.database.base import session_getter
|
||||
from bisheng.database.models.assistant import AssistantDao, AssistantStatus
|
||||
from bisheng.database.models.flow import Flow, FlowDao
|
||||
from bisheng.database.models.assistant import AssistantDao
|
||||
from bisheng.database.models.flow import Flow, FlowDao, FlowStatus, FlowType
|
||||
from bisheng.database.models.flow_version import FlowVersionDao
|
||||
from bisheng.database.models.message import ChatMessage, ChatMessageDao, ChatMessageRead
|
||||
from bisheng.database.models.mark_record import MarkRecordDao, MarkRecordStatus
|
||||
from bisheng.database.models.mark_task import MarkTaskDao
|
||||
from bisheng.database.models.message import ChatMessage, ChatMessageDao, ChatMessageRead, LikedType
|
||||
from bisheng.database.models.session import MessageSession, MessageSessionDao, SensitiveStatus
|
||||
from bisheng.database.models.user import UserDao
|
||||
from bisheng.database.models.user_group import UserGroupDao
|
||||
from bisheng.graph.graph.base import Graph
|
||||
from bisheng.utils import generate_uuid
|
||||
from bisheng.utils.logger import logger
|
||||
from bisheng.utils.util import get_cache_key
|
||||
from fastapi import APIRouter, HTTPException, Query, WebSocket, WebSocketException, status
|
||||
from fastapi.params import Depends
|
||||
from fastapi.responses import StreamingResponse
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
from sqlalchemy import func
|
||||
from sqlmodel import select
|
||||
|
||||
router = APIRouter(tags=['Chat'])
|
||||
chat_manager = ChatManager()
|
||||
@@ -33,6 +47,172 @@ flow_data_store = redis_client
|
||||
expire = 600 # reids 60s 过期
|
||||
|
||||
|
||||
@router.post('/chat/completions', response_class=StreamingResponse)
|
||||
async def chat_completions(request: APIChatCompletion, Authorize: AuthJWT = Depends()):
|
||||
# messages 为openai 格式。目前不支持openai的复杂多轮,先临时处理
|
||||
message = None
|
||||
if request.messages:
|
||||
last_message = request.messages[-1]
|
||||
if 'content' in last_message:
|
||||
message = last_message['content']
|
||||
else:
|
||||
logger.info('last_message={}', last_message)
|
||||
message = last_message
|
||||
session_id = request.session_id or generate_uuid()
|
||||
|
||||
payload = {'user_name': 'root', 'user_id': 1, 'role': 'admin'}
|
||||
access_token = Authorize.create_access_token(subject=json.dumps(payload), expires_time=864000)
|
||||
url = f'ws://127.0.0.1:7860/api/v1/chat/{request.model}?chat_id={session_id}&t={access_token}'
|
||||
web_conn = await chat_imp.get_connection(url, session_id)
|
||||
|
||||
return StreamingResponse(chat_imp.event_stream(web_conn, message, session_id, request.model,
|
||||
request.streaming),
|
||||
media_type='text/event-stream')
|
||||
|
||||
|
||||
@router.get('/chat/app/list')
|
||||
def get_app_chat_list(*,
|
||||
keyword: Optional[str] = None,
|
||||
mark_user: Optional[str] = None,
|
||||
mark_status: Optional[int] = None,
|
||||
task_id: Optional[int] = Query(default=None, description='标注任务ID'),
|
||||
flow_type: Optional[int] = None,
|
||||
page_num: Optional[int] = 1,
|
||||
page_size: Optional[int] = 20,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
""" 通过标注任务ID获取对应的会话列表 """
|
||||
|
||||
group_flow_ids = []
|
||||
flow_ids, user_ids = [], []
|
||||
|
||||
user_groups = UserGroupDao.get_user_admin_group(login_user.user_id)
|
||||
if task_id:
|
||||
if not login_user.is_admin():
|
||||
task = MarkTaskDao.get_task_byid(task_id)
|
||||
if str(login_user.user_id) not in task.process_users.split(','):
|
||||
raise HTTPException(status_code=403, detail='没有权限')
|
||||
# 判断下是否是用户组管理员
|
||||
if user_groups:
|
||||
task = MarkTaskDao.get_task_byid(task_id)
|
||||
group_flow_ids = task.app_id.split(',')
|
||||
# group_flow_ids.extend([app_id for one in t_list for app_id in one.app_id.split(",")])
|
||||
if not group_flow_ids:
|
||||
return resp_200(PageList(list=[], total=0))
|
||||
else:
|
||||
task = MarkTaskDao.get_task_byid(task_id)
|
||||
if str(login_user.user_id) not in task.process_users.split(','):
|
||||
raise HTTPException(status_code=403, detail='没有权限')
|
||||
# 普通用户
|
||||
# user_ids = [login_user.user_id]
|
||||
group_flow_ids = MarkTaskDao.get_task_byid(task_id).app_id.split(',')
|
||||
|
||||
else:
|
||||
group_flow_ids = MarkTaskDao.get_task_byid(task_id).app_id.split(',')
|
||||
|
||||
if keyword:
|
||||
flows = FlowDao.get_flow_list_by_name(name=keyword)
|
||||
assistants, _ = AssistantDao.get_all_assistants(name=keyword, page=0, limit=0)
|
||||
users = UserDao.search_user_by_name(user_name=keyword)
|
||||
if flows:
|
||||
flow_ids = [flow.id for flow in flows]
|
||||
if assistants:
|
||||
flow_ids.extend([assistant.id for assistant in assistants])
|
||||
if user_ids:
|
||||
user_ids = [user.user_id for user in users]
|
||||
# 检索内容为空
|
||||
if not flow_ids and not user_ids:
|
||||
return resp_200(PageList(list=[], total=0))
|
||||
|
||||
if group_flow_ids:
|
||||
if flow_ids and keyword:
|
||||
flow_ids = flow_ids
|
||||
else:
|
||||
flow_ids = group_flow_ids
|
||||
|
||||
# 获取会话列表
|
||||
res = MessageSessionDao.filter_session(flow_ids=flow_ids, user_ids=user_ids)
|
||||
total = len(res)
|
||||
|
||||
# 查询会话的状态
|
||||
chat_status_ids = [one.chat_id for one in res]
|
||||
chat_status_ids = MarkRecordDao.filter_records(task_id=task_id, chat_ids=chat_status_ids)
|
||||
chat_status_ids = {one.session_id: one for one in chat_status_ids}
|
||||
|
||||
result = []
|
||||
for one in res:
|
||||
tmp = AppChatList(
|
||||
chat_id=one.chat_id,
|
||||
flow_id=one.flow_id,
|
||||
flow_name=one.flow_name,
|
||||
flow_type=one.flow_type,
|
||||
user_id=one.user_id,
|
||||
user_name=one.user_id,
|
||||
create_time=one.create_time,
|
||||
like_count=one.like,
|
||||
dislike_count=one.dislike,
|
||||
copied_count=one.copied,
|
||||
mark_status=MarkRecordStatus.DEFAULT.value,
|
||||
mark_user=None,
|
||||
)
|
||||
if mark_info := chat_status_ids.get(one.chat_id):
|
||||
tmp.mark_id = mark_info.create_id
|
||||
tmp.mark_status = mark_info.status if mark_info.status is not None else 1
|
||||
tmp.mark_user = mark_info.create_user
|
||||
if mark_status:
|
||||
if mark_status != tmp.mark_status:
|
||||
continue
|
||||
if mark_user:
|
||||
users = [int(one) for one in mark_user.split(',')]
|
||||
if tmp.mark_id not in users:
|
||||
continue
|
||||
result.append(tmp)
|
||||
|
||||
result = result[(page_num - 1) * page_size: page_num * page_size]
|
||||
|
||||
return resp_200(PageList(list=result, total=total))
|
||||
|
||||
|
||||
@router.get('/chat/history')
|
||||
def get_chatmessage(*,
|
||||
chat_id: str,
|
||||
flow_id: str,
|
||||
id: Optional[str] = None,
|
||||
page_size: Optional[int] = 20,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
if not chat_id or not flow_id:
|
||||
return {'code': 500, 'message': 'chat_id 和 flow_id 必传参数'}
|
||||
where = select(ChatMessage).where(ChatMessage.flow_id == flow_id,
|
||||
ChatMessage.chat_id == chat_id)
|
||||
if id:
|
||||
where = where.where(ChatMessage.id < int(id))
|
||||
with session_getter() as session:
|
||||
db_message = session.exec(where.order_by(ChatMessage.id.desc()).limit(page_size)).all()
|
||||
return resp_200(db_message)
|
||||
|
||||
|
||||
@router.post('/chat/conversation/rename')
|
||||
def rename(conversationId: str = Body(..., description='会话id', embed=True),
|
||||
name: str = Body(..., description='会话名称', embed=True),
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
conversation = MessageSessionDao.get_one(conversationId)
|
||||
conversation.flow_name = name
|
||||
MessageSessionDao.insert_one(conversation)
|
||||
return resp_200()
|
||||
|
||||
|
||||
@router.post('/chat/conversation/copy')
|
||||
def copy(conversationId: str = Body(..., description='会话id', embed=True), ):
|
||||
conversation = MessageSessionDao.get_one(conversationId)
|
||||
conversation.chat_id = uuid4().hex
|
||||
conversation = MessageSessionDao.insert_one(conversation)
|
||||
msg_list = ChatMessageDao.get_messages_by_chat_id(conversationId)
|
||||
if msg_list:
|
||||
for msg in msg_list:
|
||||
msg.chat_id = conversation.chat_id
|
||||
msg.id = None
|
||||
ChatMessageDao.insert_one(msg)
|
||||
|
||||
|
||||
@router.get('/chat/history',
|
||||
response_model=UnifiedResponseModel[List[ChatMessageRead]],
|
||||
status_code=200)
|
||||
@@ -41,14 +221,11 @@ def get_chatmessage(*,
|
||||
flow_id: str,
|
||||
id: Optional[str] = None,
|
||||
page_size: Optional[int] = 20,
|
||||
Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
payload = json.loads(Authorize.get_jwt_subject())
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
if not chat_id or not flow_id:
|
||||
return {'code': 500, 'message': 'chat_id 和 flow_id 必传参数'}
|
||||
where = select(ChatMessage).where(ChatMessage.flow_id == flow_id,
|
||||
ChatMessage.chat_id == chat_id,
|
||||
ChatMessage.user_id == payload.get('user_id'))
|
||||
ChatMessage.chat_id == chat_id)
|
||||
if id:
|
||||
where = where.where(ChatMessage.id < int(id))
|
||||
with session_getter() as session:
|
||||
@@ -57,138 +234,259 @@ def get_chatmessage(*,
|
||||
|
||||
|
||||
@router.delete('/chat/{chat_id}', status_code=200)
|
||||
def del_chat_id(*, chat_id: str, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
payload = json.loads(Authorize.get_jwt_subject())
|
||||
def del_chat_id(*,
|
||||
request: Request,
|
||||
chat_id: str,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
# 获取一条消息
|
||||
message = ChatMessageDao.get_latest_message_by_chatid(chat_id)
|
||||
if message:
|
||||
# 处理临时数据
|
||||
col_name = f'tmp_{message.flow_id.hex}_{chat_id}'
|
||||
logger.info('tmp_delete_milvus col={}', col_name)
|
||||
delete_vector(col_name, None)
|
||||
delete_es(col_name)
|
||||
ChatMessageDao.delete_by_user_chat_id(payload.get('user_id'), chat_id)
|
||||
session_chat = MessageSessionDao.get_one(chat_id)
|
||||
|
||||
if not session_chat or session_chat.is_delete:
|
||||
return resp_200(message='删除成功')
|
||||
# 处理临时数据
|
||||
col_name = f'tmp_{session_chat.flow_id}_{chat_id}'
|
||||
logger.info('tmp_delete_milvus col={}', col_name)
|
||||
delete_vector(col_name, None)
|
||||
delete_es(col_name)
|
||||
if session_chat.flow_type == FlowType.ASSISTANT.value:
|
||||
assistant_info = AssistantDao.get_one_assistant(session_chat.flow_id)
|
||||
if assistant_info:
|
||||
AuditLogService.delete_chat_assistant(login_user, get_request_ip(request), assistant_info)
|
||||
else:
|
||||
# 判断下是助手还是技能, 写审计日志
|
||||
flow_info = FlowDao.get_flow_by_id(session_chat.flow_id)
|
||||
if flow_info and flow_info.flow_type == FlowType.FLOW.value:
|
||||
AuditLogService.delete_chat_flow(login_user, get_request_ip(request), flow_info)
|
||||
elif flow_info:
|
||||
AuditLogService.delete_chat_workflow(login_user, get_request_ip(request), flow_info)
|
||||
|
||||
# 设置会话的删除状态
|
||||
MessageSessionDao.delete_session(chat_id)
|
||||
|
||||
return resp_200(message='删除成功')
|
||||
|
||||
|
||||
@router.post('/chat/message', status_code=200)
|
||||
def add_chat_messages(*,
|
||||
request: Request,
|
||||
data: AddChatMessages,
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
"""
|
||||
添加一条完整问答记录, 安全检查写入使用
|
||||
"""
|
||||
logger.debug(f'gateway add_chat_messages {data}')
|
||||
flow_id = data.flow_id
|
||||
chat_id = data.chat_id
|
||||
if not chat_id or not flow_id:
|
||||
raise HTTPException(status_code=500, detail='chat_id 和 flow_id 必传参数')
|
||||
save_human_message = data.human_message
|
||||
flow_info = FlowDao.get_flow_by_id(flow_id)
|
||||
if flow_info and flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
# 工作流的输入,需要从输入里解析出来实际的输入内容
|
||||
try:
|
||||
tmp_human_message = json.loads(data.human_message)
|
||||
for node_id, node_input in tmp_human_message.items():
|
||||
save_human_message = node_input.get('message')
|
||||
except:
|
||||
save_human_message = data.human_message
|
||||
|
||||
human_message = ChatMessage(flow_id=flow_id,
|
||||
chat_id=chat_id,
|
||||
user_id=login_user.user_id,
|
||||
is_bot=False,
|
||||
message=save_human_message,
|
||||
sensitive_status=SensitiveStatus.VIOLATIONS.value,
|
||||
type='human',
|
||||
category='question')
|
||||
bot_message = ChatMessage(flow_id=flow_id,
|
||||
chat_id=chat_id,
|
||||
user_id=login_user.user_id,
|
||||
is_bot=True,
|
||||
message=data.answer_message,
|
||||
sensitive_status=SensitiveStatus.PASS.value,
|
||||
type='bot',
|
||||
category='answer')
|
||||
message_dbs = ChatMessageDao.insert_batch([human_message, bot_message])
|
||||
# 更新会话的状态
|
||||
MessageSessionDao.update_sensitive_status(chat_id, SensitiveStatus.VIOLATIONS)
|
||||
|
||||
# 写审计日志, 判断是否是新建会话
|
||||
session_info = MessageSessionDao.get_one(chat_id=chat_id)
|
||||
if not session_info:
|
||||
# 新建会话
|
||||
# 判断下是助手还是技能, 写审计日志
|
||||
if flow_info:
|
||||
MessageSessionDao.insert_one(MessageSession(
|
||||
chat_id=chat_id,
|
||||
flow_id=flow_id,
|
||||
flow_type=flow_info.flow_type,
|
||||
flow_name=flow_info.name,
|
||||
user_id=login_user.user_id,
|
||||
sensitive_status=SensitiveStatus.VIOLATIONS.value,
|
||||
))
|
||||
if flow_info.flow_type == FlowType.FLOW.value:
|
||||
AuditLogService.create_chat_flow(login_user, get_request_ip(request), flow_id, flow_info)
|
||||
elif flow_info.flow_type == FlowType.WORKFLOW.value:
|
||||
AuditLogService.create_chat_workflow(login_user, get_request_ip(request), flow_id, flow_info)
|
||||
else:
|
||||
assistant_info = AssistantDao.get_one_assistant(flow_id)
|
||||
if assistant_info:
|
||||
MessageSessionDao.insert_one(MessageSession(
|
||||
chat_id=chat_id,
|
||||
flow_id=flow_id,
|
||||
flow_type=FlowType.ASSISTANT.value,
|
||||
flow_name=assistant_info.name,
|
||||
user_id=login_user.user_id,
|
||||
sensitive_status=SensitiveStatus.VIOLATIONS.value,
|
||||
))
|
||||
AuditLogService.create_chat_assistant(login_user, get_request_ip(request),
|
||||
flow_id)
|
||||
|
||||
return resp_200(data=message_dbs, message='添加成功')
|
||||
|
||||
|
||||
@router.put('/chat/message/{message_id}', status_code=200)
|
||||
def update_chat_message(*,
|
||||
message_id: int,
|
||||
message: str = Body(embed=True),
|
||||
category: str = Body(default=None, embed=True),
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
""" 更新一条消息的内容 安全检查使用"""
|
||||
logger.info(
|
||||
f'update_chat_message message_id={message_id} message={message} login_user={login_user.user_name}'
|
||||
)
|
||||
chat_message = ChatMessageDao.get_message_by_id(message_id)
|
||||
if not chat_message:
|
||||
return resp_200(message='消息不存在')
|
||||
if chat_message.user_id != login_user.user_id:
|
||||
return resp_200(message='用户不一致')
|
||||
|
||||
chat_message.message = message
|
||||
if category:
|
||||
chat_message.category = category
|
||||
chat_message.source = False
|
||||
chat_message.sensitive_status = SensitiveStatus.VIOLATIONS.value
|
||||
|
||||
ChatMessageDao.update_message_model(chat_message)
|
||||
|
||||
MessageSessionDao.update_sensitive_status(chat_message.chat_id, SensitiveStatus.VIOLATIONS)
|
||||
|
||||
return resp_200(message='更新成功')
|
||||
|
||||
|
||||
@router.delete('/chat/message/{message_id}', status_code=200)
|
||||
def del_message_id(*, message_id: str, login_user: UserPayload = Depends(get_login_user)):
|
||||
ChatMessageDao.delete_by_message_id(login_user.user_id, message_id)
|
||||
|
||||
return resp_200(message='删除成功')
|
||||
|
||||
|
||||
@router.post('/liked', status_code=200)
|
||||
def like_response(*, data: ChatInput, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
payload = json.loads(Authorize.get_jwt_subject())
|
||||
def like_response(*, data: ChatInput):
|
||||
message_id = data.message_id
|
||||
liked = data.liked
|
||||
with session_getter() as session:
|
||||
message = session.get(ChatMessage, message_id)
|
||||
if message:
|
||||
logger.info('act=add_liked user_id={} liked={}', payload.get('user_id'), liked)
|
||||
message.liked = liked
|
||||
with session_getter() as session:
|
||||
session.add(message)
|
||||
session.commit()
|
||||
logger.info('k=s act=liked message_id={} liked={}', message_id, liked)
|
||||
message = ChatMessageDao.get_message_by_id(data.message_id)
|
||||
if not message:
|
||||
raise NotFoundError.http_exception()
|
||||
|
||||
if message.liked == data.liked:
|
||||
return resp_200(message='操作成功')
|
||||
|
||||
like_count = 0
|
||||
dislike_count = 0
|
||||
if message.liked == LikedType.UNRATED.value:
|
||||
if data.liked == LikedType.LIKED.value:
|
||||
like_count = 1
|
||||
elif data.liked == LikedType.DISLIKED.value:
|
||||
dislike_count = 1
|
||||
elif message.liked == LikedType.LIKED.value:
|
||||
like_count = -1
|
||||
if data.liked == LikedType.DISLIKED.value:
|
||||
dislike_count = 1
|
||||
elif message.liked == LikedType.DISLIKED.value:
|
||||
dislike_count = -1
|
||||
if data.liked == LikedType.LIKED.value:
|
||||
like_count = 1
|
||||
|
||||
message.liked = data.liked
|
||||
ChatMessageDao.update_message_model(message)
|
||||
logger.info('k=s act=liked message_id={} liked={}', message_id, data.liked)
|
||||
|
||||
# 更新会话表的点赞点踩数
|
||||
MessageSessionDao.add_like_count(message.chat_id, like_count)
|
||||
MessageSessionDao.add_dislike_count(message.chat_id, dislike_count)
|
||||
|
||||
return resp_200(message='操作成功')
|
||||
|
||||
|
||||
@router.post('/chat/copied', status_code=200)
|
||||
def copied_message(message_id: int = Body(embed=True)):
|
||||
""" 上传复制message的数据 """
|
||||
message = ChatMessageDao.get_message_by_id(message_id)
|
||||
if not message:
|
||||
raise NotFoundError.http_exception()
|
||||
if message.copied != 1:
|
||||
ChatMessageDao.update_message_copied(message_id, 1)
|
||||
MessageSessionDao.add_copied_count(message.chat_id, 1)
|
||||
return resp_200(message='操作成功')
|
||||
|
||||
|
||||
@router.post('/chat/comment', status_code=200)
|
||||
def comment_resp(*, data: ChatInput, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
def comment_resp(*, data: ChatInput):
|
||||
comment_answer(data.message_id, data.comment)
|
||||
return resp_200(message='操作成功')
|
||||
|
||||
|
||||
@router.get('/chat/list', response_model=UnifiedResponseModel[List[ChatList]], status_code=200)
|
||||
def get_chatlist_list(*, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
payload = json.loads(Authorize.get_jwt_subject())
|
||||
|
||||
smt = (select(ChatMessage.flow_id, ChatMessage.chat_id,
|
||||
func.max(ChatMessage.create_time).label('create_time'),
|
||||
func.max(ChatMessage.update_time).label('update_time')).where(
|
||||
ChatMessage.user_id == payload.get('user_id')).group_by(
|
||||
ChatMessage.flow_id,
|
||||
ChatMessage.chat_id).order_by(func.max(ChatMessage.create_time).desc()))
|
||||
with session_getter() as session:
|
||||
db_message = session.exec(smt).all()
|
||||
flow_ids = [message.flow_id for message in db_message]
|
||||
with session_getter() as session:
|
||||
db_flow = session.exec(select(Flow).where(Flow.id.in_(flow_ids))).all()
|
||||
|
||||
assistant_chats = AssistantDao.get_assistants_by_ids(flow_ids)
|
||||
assistant_dict = {assistant.id: assistant for assistant in assistant_chats}
|
||||
# set object
|
||||
chat_list = []
|
||||
flow_dict = {flow.id: flow for flow in db_flow}
|
||||
for i, message in enumerate(db_message):
|
||||
if message.flow_id in flow_dict:
|
||||
chat_list.append(
|
||||
ChatList(flow_name=flow_dict[message.flow_id].name,
|
||||
flow_description=flow_dict[message.flow_id].description,
|
||||
flow_id=message.flow_id,
|
||||
flow_type='flow',
|
||||
chat_id=message.chat_id,
|
||||
create_time=message.create_time,
|
||||
update_time=message.update_time))
|
||||
elif message.flow_id in assistant_dict:
|
||||
chat_list.append(
|
||||
ChatList(flow_name=assistant_dict[message.flow_id].name,
|
||||
flow_description=assistant_dict[message.flow_id].desc,
|
||||
flow_id=message.flow_id,
|
||||
chat_id=message.chat_id,
|
||||
flow_type='assistant',
|
||||
create_time=message.create_time,
|
||||
update_time=message.update_time))
|
||||
else:
|
||||
# 通过接口创建的会话记录,不关联技能或者助手
|
||||
logger.debug(f'unknown message.flow_id={message.flow_id}')
|
||||
return resp_200(chat_list)
|
||||
@router.get('/chat/list')
|
||||
def get_session_list(page: Optional[int] = Query(default=1, ge=1, le=1000),
|
||||
limit: Optional[int] = Query(default=10, ge=1, le=100),
|
||||
flow_type: Optional[List[int]] = Query(default=None, description='技能类型'),
|
||||
login_user: UserPayload = Depends(get_login_user)):
|
||||
res = MessageSessionDao.filter_session(user_ids=[login_user.user_id],
|
||||
flow_type=flow_type,
|
||||
page=page,
|
||||
limit=limit,
|
||||
include_delete=False)
|
||||
chat_ids = []
|
||||
flow_ids = []
|
||||
for one in res:
|
||||
chat_ids.append(one.chat_id)
|
||||
flow_ids.append(one.flow_id)
|
||||
flow_list = FlowDao.get_flow_by_ids(flow_ids)
|
||||
assistant_list = AssistantDao.get_assistants_by_ids(flow_ids)
|
||||
logo_map = {one.id: BaseService.get_logo_share_link(one.logo) for one in flow_list}
|
||||
logo_map.update({one.id: BaseService.get_logo_share_link(one.logo) for one in assistant_list})
|
||||
latest_messages = ChatMessageDao.get_latest_message_by_chat_ids(chat_ids,
|
||||
exclude_category=WorkflowEventType.UserInput.value)
|
||||
latest_messages = {one.chat_id: one for one in latest_messages}
|
||||
return resp_200([
|
||||
ChatList(
|
||||
chat_id=one.chat_id,
|
||||
flow_id=one.flow_id,
|
||||
flow_name=one.flow_name,
|
||||
flow_type=one.flow_type,
|
||||
logo=logo_map.get(one.flow_id, ''),
|
||||
latest_message=latest_messages.get(one.chat_id, None),
|
||||
create_time=one.create_time,
|
||||
update_time=one.update_time) for one in res
|
||||
])
|
||||
|
||||
|
||||
# 获取所有已上线的技能和助手
|
||||
@router.get('/chat/online',
|
||||
response_model=UnifiedResponseModel[List[FlowGptsOnlineList]],
|
||||
status_code=200)
|
||||
def get_online_chat(*, Authorize: AuthJWT = Depends()):
|
||||
Authorize.jwt_required()
|
||||
payload = json.loads(Authorize.get_jwt_subject())
|
||||
user = UserPayload(**payload)
|
||||
user_id = user.user_id
|
||||
res = []
|
||||
# 获取所有已上线的助手
|
||||
if user.is_admin():
|
||||
all_assistant = AssistantDao.get_all_online_assistants()
|
||||
flows = FlowDao.get_all_online_flows()
|
||||
else:
|
||||
assistants = AssistantService.get_assistant(user, None, AssistantStatus.ONLINE.value, 0, 0)
|
||||
all_assistant = assistants.data.get('data')
|
||||
flows = FlowDao.get_user_access_online_flows(user_id)
|
||||
for one in all_assistant:
|
||||
res.append(
|
||||
FlowGptsOnlineList(id=one.id.hex,
|
||||
name=one.name,
|
||||
desc=one.desc,
|
||||
create_time=one.create_time,
|
||||
update_time=one.update_time,
|
||||
flow_type='assistant'))
|
||||
|
||||
# 获取用户可见的所有已上线的技能
|
||||
for one in flows:
|
||||
res.append(
|
||||
FlowGptsOnlineList(id=one.id.hex,
|
||||
name=one.name,
|
||||
desc=one.description,
|
||||
create_time=one.create_time,
|
||||
update_time=one.update_time,
|
||||
flow_type='flow'))
|
||||
res.sort(key=lambda x: x.update_time, reverse=True)
|
||||
return resp_200(data=res)
|
||||
@router.get('/chat/online')
|
||||
def get_online_chat(*,
|
||||
keyword: Optional[str] = None,
|
||||
tag_id: Optional[int] = None,
|
||||
page: Optional[int] = 1,
|
||||
limit: Optional[int] = 10,
|
||||
user: UserPayload = Depends(get_login_user)):
|
||||
data, _ = WorkFlowService.get_all_flows(user, keyword, FlowStatus.ONLINE.value, tag_id, None, page, limit)
|
||||
return resp_200(data=data)
|
||||
|
||||
|
||||
@router.websocket('/chat/{flow_id}')
|
||||
async def chat(
|
||||
*,
|
||||
flow_id: str,
|
||||
flow_id: UUID,
|
||||
websocket: WebSocket,
|
||||
t: Optional[str] = None,
|
||||
chat_id: Optional[str] = None,
|
||||
@@ -196,16 +494,15 @@ async def chat(
|
||||
Authorize: AuthJWT = Depends(),
|
||||
):
|
||||
"""Websocket endpoint for chat."""
|
||||
flow_id = flow_id.hex
|
||||
try:
|
||||
if t:
|
||||
Authorize.jwt_required(auth_from='websocket', token=t)
|
||||
Authorize._token = t
|
||||
else:
|
||||
Authorize.jwt_required(auth_from='websocket', websocket=websocket)
|
||||
|
||||
payload = Authorize.get_jwt_subject()
|
||||
payload = json.loads(payload)
|
||||
user_id = payload.get('user_id')
|
||||
login_user = await get_login_user(Authorize)
|
||||
user_id = login_user.user_id
|
||||
if chat_id:
|
||||
with session_getter() as session:
|
||||
db_flow = session.get(Flow, flow_id)
|
||||
@@ -254,9 +551,7 @@ async def chat(
|
||||
await websocket.close(code=status.WS_1011_INTERNAL_ERROR, reason=messsage)
|
||||
|
||||
|
||||
@router.post('/build/init/{flow_id}',
|
||||
response_model=UnifiedResponseModel[InitResponse],
|
||||
status_code=201)
|
||||
@router.post('/build/init/{flow_id}')
|
||||
async def init_build(*,
|
||||
graph_data: dict,
|
||||
flow_id: str,
|
||||
@@ -267,7 +562,7 @@ async def init_build(*,
|
||||
|
||||
if chat_id:
|
||||
with session_getter() as session:
|
||||
graph_data = session.get(Flow, UUID(flow_id).hex).data
|
||||
graph_data = session.get(Flow, flow_id).data
|
||||
elif version_id:
|
||||
flow_data_key = flow_data_key + '_' + str(version_id)
|
||||
graph_data = FlowVersionDao.get_version_by_id(version_id).data
|
||||
@@ -280,7 +575,7 @@ async def init_build(*,
|
||||
|
||||
# Delete from cache if already exists
|
||||
flow_data_store.hset(flow_data_key,
|
||||
map={
|
||||
mapping={
|
||||
'graph_data': json.dumps(graph_data),
|
||||
'status': BuildStatus.STARTED.value
|
||||
},
|
||||
@@ -347,7 +642,7 @@ async def stream_build(flow_id: str,
|
||||
async for message in build_flow(graph_data=graph_data,
|
||||
artifacts=artifacts,
|
||||
process_file=False,
|
||||
flow_id=UUID(flow_id).hex,
|
||||
flow_id=flow_id,
|
||||
chat_id=chat_id):
|
||||
if isinstance(message, Graph):
|
||||
graph = message
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
import json
|
||||
from typing import List
|
||||
|
||||
from fastapi import APIRouter, Body, Depends
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
|
||||
from bisheng import __version__
|
||||
from bisheng.api.services.component import ComponentService
|
||||
from bisheng.api.services.user_service import get_login_user
|
||||
from bisheng.api.utils import update_frontend_node_with_template_values
|
||||
from bisheng.api.v1.schemas import (CreateComponentReq, CustomComponentCode, UnifiedResponseModel,
|
||||
resp_200, resp_500)
|
||||
@@ -10,13 +14,11 @@ from bisheng.database.models.component import Component
|
||||
from bisheng.interface.custom import CustomComponent
|
||||
from bisheng.interface.custom.directory_reader import DirectoryReader
|
||||
from bisheng.interface.custom.utils import build_custom_component_template
|
||||
from fastapi import APIRouter, Body, Depends
|
||||
from fastapi_jwt_auth import AuthJWT
|
||||
|
||||
router = APIRouter(prefix='/component', tags=['Component'])
|
||||
router = APIRouter(prefix='/component', tags=['Component'], dependencies=[Depends(get_login_user)])
|
||||
|
||||
|
||||
@router.get('', response_model=UnifiedResponseModel[List[Component]])
|
||||
@router.get('')
|
||||
def get_all_components(*, Authorize: AuthJWT = Depends()):
|
||||
# get login user
|
||||
Authorize.jwt_required()
|
||||
@@ -24,7 +26,7 @@ def get_all_components(*, Authorize: AuthJWT = Depends()):
|
||||
return ComponentService.get_all_component(current_user)
|
||||
|
||||
|
||||
@router.post('', response_model=UnifiedResponseModel[Component])
|
||||
@router.post('')
|
||||
def save_components(*, data: CreateComponentReq, Authorize: AuthJWT = Depends()):
|
||||
# get login user
|
||||
Authorize.jwt_required()
|
||||
@@ -36,7 +38,7 @@ def save_components(*, data: CreateComponentReq, Authorize: AuthJWT = Depends())
|
||||
return ComponentService.save_component(component)
|
||||
|
||||
|
||||
@router.patch('', response_model=UnifiedResponseModel[Component])
|
||||
@router.patch('')
|
||||
def update_component(*, data: CreateComponentReq, Authorize: AuthJWT = Depends()):
|
||||
# get login user
|
||||
Authorize.jwt_required()
|
||||
@@ -48,7 +50,7 @@ def update_component(*, data: CreateComponentReq, Authorize: AuthJWT = Depends()
|
||||
return ComponentService.update_component(component)
|
||||
|
||||
|
||||
@router.delete('', response_model=UnifiedResponseModel[Component])
|
||||
@router.delete('')
|
||||
def delete_component(*,
|
||||
name: str = Body(..., embed=True, description='组件名'),
|
||||
Authorize: AuthJWT = Depends()):
|
||||
@@ -58,7 +60,7 @@ def delete_component(*,
|
||||
return ComponentService.delete_component(current_user.get('user_id'), name)
|
||||
|
||||
|
||||
@router.post('/custom_component', response_model=UnifiedResponseModel[Component])
|
||||
@router.post('/custom_component')
|
||||
async def custom_component(
|
||||
raw_code: CustomComponentCode,
|
||||
Authorize: AuthJWT = Depends(),
|
||||
@@ -76,7 +78,7 @@ async def custom_component(
|
||||
return resp_200(data=built_frontend_node)
|
||||
|
||||
|
||||
@router.post('/custom_component/reload', response_model=UnifiedResponseModel[Component])
|
||||
@router.post('/custom_component/reload')
|
||||
async def reload_custom_component(path: str, Authorize: AuthJWT = Depends()):
|
||||
from bisheng.interface.custom.utils import build_custom_component_template
|
||||
|
||||
@@ -96,7 +98,7 @@ async def reload_custom_component(path: str, Authorize: AuthJWT = Depends()):
|
||||
return resp_500(message=str(exc))
|
||||
|
||||
|
||||
@router.post('/custom_component/update', response_model=UnifiedResponseModel[Component])
|
||||
@router.post('/custom_component/update')
|
||||
async def custom_component_update(
|
||||
raw_code: CustomComponentCode,
|
||||
Authorize: AuthJWT = Depends(),
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
from typing import List
|
||||
|
||||
from bisheng.api.services.dataset_service import DatasetService
|
||||
from bisheng.api.services.user_service import UserPayload, get_login_user
|
||||
from bisheng.api.v1.schema.dataset_param import CreateDatasetParam
|
||||
from bisheng.api.v1.schemas import UnifiedResponseModel, resp_200
|
||||
from bisheng.database.models.dataset import DatasetRead
|
||||
from fastapi import APIRouter, Depends, Request
|
||||
|
||||
# build router
|
||||
router = APIRouter(prefix='/dataset', tags=['FineTune'])
|
||||
|
||||
|
||||
@router.get('/list', summary='获取数据集列表')
|
||||
def list_dataset(*,
|
||||
keyword: str = None,
|
||||
page: int = 1,
|
||||
limit: int = 10) -> UnifiedResponseModel[List[DatasetRead]]:
|
||||
"""
|
||||
获取数据集列表
|
||||
"""
|
||||
res, count = DatasetService.build_dataset_list(page, limit, keyword)
|
||||
return resp_200(data={'list': res, 'total': count})
|
||||
|
||||
|
||||
@router.post('/create', summary='创建数据集')
|
||||
def create_dataset(
|
||||
*,
|
||||
request: Request,
|
||||
data: CreateDatasetParam,
|
||||
login_user: UserPayload = Depends(get_login_user),
|
||||
) -> UnifiedResponseModel:
|
||||
"""
|
||||
创建数据集
|
||||
"""
|
||||
dataset = DatasetService.create_dataset(login_user.user_id, data)
|
||||
return resp_200(data=dataset)
|
||||
|
||||
|
||||
@router.delete('/del', summary='删除数据集')
|
||||
def delete_dataset(
|
||||
*,
|
||||
request: Request,
|
||||
dataset_id: int,
|
||||
login_user: UserPayload = Depends(get_login_user),
|
||||
) -> UnifiedResponseModel:
|
||||
"""
|
||||
创建数据集
|
||||
"""
|
||||
DatasetService.delete_dataset(dataset_id)
|
||||
return resp_200()
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user