@poetic-ai/poetic 1.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/SECURITY.md +47 -0
- package/.nvmrc +1 -0
- package/.poetic/README.md +37 -0
- package/.poetic/providers/catalog.json +12051 -0
- package/.poetic/providers/pricing.json +1272 -0
- package/.poetic/providers/registry.json +3369 -0
- package/CHANGELOG.md +552 -0
- package/CODE_OF_CONDUCT.md +40 -0
- package/CONTRIBUTING.md +23 -0
- package/INSTALL.md +454 -0
- package/LICENSE +21 -0
- package/README.md +474 -0
- package/dist/BasicOptimizationCompetitionRunner-5H5TBYO4.js +153 -0
- package/dist/ExecutionTracker-UVONC4C6.js +16 -0
- package/dist/OptimizationConfig-7DFY2TST.js +17 -0
- package/dist/OptimizationEngine-JIMGP3LY.js +1597 -0
- package/dist/PromptStore-U6ZD3ZMQ.js +18 -0
- package/dist/actions-LUJPINLJ.js +322 -0
- package/dist/agent-ingest-3DJUJ7FU.js +53 -0
- package/dist/aggregator-DNCBINFO.js +12 -0
- package/dist/ai-judge-5QT5QJVU.js +156 -0
- package/dist/allowlist-grounding-RO6HKUFI.js +16 -0
- package/dist/allowlist-utils-23V7TJ3G.js +28 -0
- package/dist/anchored-turn-service-JULUCUIK.js +489 -0
- package/dist/api-transport-4TYTNLTJ.js +1087 -0
- package/dist/apply-completion-mode-AW2MOS3R.js +89 -0
- package/dist/artifact-migrator-23L45ISD.js +267 -0
- package/dist/ask-HWVV6QFR.js +135 -0
- package/dist/auth-5KEDKAPP.js +67 -0
- package/dist/auth-7TT3JINV.js +78 -0
- package/dist/auth-E6XUNZ5M.js +68 -0
- package/dist/auth-H3XDZE22.js +134 -0
- package/dist/auth-OC4E6633.js +71 -0
- package/dist/auth-S4CTX4GQ.js +44 -0
- package/dist/auth-YLH3S7EY.js +69 -0
- package/dist/auth-liveness-TDDEZY3T.js +46 -0
- package/dist/auto-optimizer-VPRWOWGU.js +80 -0
- package/dist/autoloop-cleanup-JOGKTLWA.js +109 -0
- package/dist/autoloop-evidence-gates-PCDALGVJ.js +38 -0
- package/dist/autoloop-gate-policy-Z7FVJVXV.js +175 -0
- package/dist/autoloop-helpers-2KTK43GL.js +22 -0
- package/dist/autoloop-ledger-W7QO62PX.js +96 -0
- package/dist/autoloop-ledger-subscriber-Z5RW5EIE.js +162 -0
- package/dist/autoloop-run-defaults-FERGRZGP.js +19 -0
- package/dist/autoloop-run-lock-CT2NWSV3.js +197 -0
- package/dist/autoloop-success-J2VKKBXQ.js +85 -0
- package/dist/autoloop-termination-summary-RN75F422.js +58 -0
- package/dist/autoloop-trajectory-XFIEXP5Z.js +282 -0
- package/dist/autoloop-wall-clock-T7ZVIQOC.js +12 -0
- package/dist/autonomous-loop-controller-PW4XVFGQ.js +4120 -0
- package/dist/backend-P3DAYRP7.js +29 -0
- package/dist/background-executor-CBN552V4.js +257 -0
- package/dist/backlog-execution-intent-U4XYPTV6.js +15 -0
- package/dist/backup-active-store-5WXG6SXT.js +17 -0
- package/dist/backup-telemetry-db-XGLXAA3N.js +254 -0
- package/dist/basic-competition-master-VCSBJVRN.js +344 -0
- package/dist/branch-archive-manager-RP6FP3ZC.js +254 -0
- package/dist/branch-cleanup-manager-RSHAI7YR.js +20 -0
- package/dist/build-gate-YHGRB4QB.js +71 -0
- package/dist/change-summary-WY7J5TOX.js +33 -0
- package/dist/chunk-23A26CSN.js +123 -0
- package/dist/chunk-27UTJD3M.js +265 -0
- package/dist/chunk-2A5CILN3.js +310 -0
- package/dist/chunk-2AQUWJPW.js +225 -0
- package/dist/chunk-2BZ53J6J.js +2344 -0
- package/dist/chunk-2DY5KNVM.js +276 -0
- package/dist/chunk-2EIA4RBE.js +659 -0
- package/dist/chunk-2IUXIGQD.js +1064 -0
- package/dist/chunk-2N42MS2Z.js +2193 -0
- package/dist/chunk-2QJ3H3L7.js +42 -0
- package/dist/chunk-2QLYTVYX.js +791 -0
- package/dist/chunk-2VSQMRCB.js +18 -0
- package/dist/chunk-2XQXYBM2.js +130 -0
- package/dist/chunk-2YFUIER7.js +426 -0
- package/dist/chunk-34CNP2HB.js +148 -0
- package/dist/chunk-35P3W3JX.js +456 -0
- package/dist/chunk-36PEBZAF.js +18 -0
- package/dist/chunk-3AOKYKA7.js +22 -0
- package/dist/chunk-3BCUNHKF.js +475 -0
- package/dist/chunk-3FLXLLB7.js +26 -0
- package/dist/chunk-3FOZ2VHV.js +100 -0
- package/dist/chunk-3G6FZCRF.js +106 -0
- package/dist/chunk-3M57DADF.js +44 -0
- package/dist/chunk-3S62LLWJ.js +543 -0
- package/dist/chunk-3ZPQLGV7.js +2904 -0
- package/dist/chunk-44N5WZZH.js +45 -0
- package/dist/chunk-45FPSA3B.js +125 -0
- package/dist/chunk-4BJ3O4RQ.js +929 -0
- package/dist/chunk-4BOP7T6L.js +2321 -0
- package/dist/chunk-4BTDUT2S.js +42 -0
- package/dist/chunk-4CSS7WVI.js +143 -0
- package/dist/chunk-4CUY7AFY.js +247 -0
- package/dist/chunk-4GSN5O5J.js +62 -0
- package/dist/chunk-4HN6DF7V.js +1027 -0
- package/dist/chunk-4IAPOG5V.js +415 -0
- package/dist/chunk-4IYIZQWV.js +48 -0
- package/dist/chunk-4NXPXE62.js +84 -0
- package/dist/chunk-4RCCF5BR.js +712 -0
- package/dist/chunk-4V7NAQ4R.js +19527 -0
- package/dist/chunk-4VJITV4P.js +445 -0
- package/dist/chunk-4YX7J6QL.js +1859 -0
- package/dist/chunk-527OBWQB.js +338 -0
- package/dist/chunk-5D4A7E3D.js +448 -0
- package/dist/chunk-5DLKYTQX.js +4003 -0
- package/dist/chunk-5EERLVZB.js +139 -0
- package/dist/chunk-5H72XE4H.js +256 -0
- package/dist/chunk-5HDP7XZG.js +91 -0
- package/dist/chunk-5LAMRH4X.js +969 -0
- package/dist/chunk-5LTN43AS.js +1972 -0
- package/dist/chunk-5RBHQYDB.js +2206 -0
- package/dist/chunk-5WCQRDRB.js +588 -0
- package/dist/chunk-5WRLK5KW.js +700 -0
- package/dist/chunk-5ZCXC23G.js +595 -0
- package/dist/chunk-6BM72ZMG.js +31 -0
- package/dist/chunk-6EDAJH2I.js +68 -0
- package/dist/chunk-6EQKFDBH.js +457 -0
- package/dist/chunk-6MFRHTRI.js +115 -0
- package/dist/chunk-6TZJRKNW.js +321 -0
- package/dist/chunk-6V7HGQFZ.js +20 -0
- package/dist/chunk-6VWVO2TU.js +3746 -0
- package/dist/chunk-6Y5TWI7H.js +146 -0
- package/dist/chunk-6Y5U7UFY.js +91 -0
- package/dist/chunk-6ZLAK3XG.js +14 -0
- package/dist/chunk-72XCRQED.js +34 -0
- package/dist/chunk-72YTZMAM.js +589 -0
- package/dist/chunk-73BBJOKA.js +263 -0
- package/dist/chunk-73NCTWKL.js +188 -0
- package/dist/chunk-77I6G4CO.js +1 -0
- package/dist/chunk-7DHRDWFS.js +3104 -0
- package/dist/chunk-7DNSNKJV.js +78 -0
- package/dist/chunk-7FMJVYBS.js +307 -0
- package/dist/chunk-7GDRQBQC.js +419 -0
- package/dist/chunk-7GPCEFUV.js +270 -0
- package/dist/chunk-7IL4A2PP.js +948 -0
- package/dist/chunk-7MMLMXO6.js +3679 -0
- package/dist/chunk-7QCZSLBH.js +1 -0
- package/dist/chunk-7R2WLDOS.js +62 -0
- package/dist/chunk-7T7RUJI7.js +949 -0
- package/dist/chunk-A2WKSVX7.js +43 -0
- package/dist/chunk-A3YGID55.js +133 -0
- package/dist/chunk-A722DEVA.js +14 -0
- package/dist/chunk-ACDIT7OP.js +1376 -0
- package/dist/chunk-ACKIPAOJ.js +302 -0
- package/dist/chunk-APPNGGN7.js +365 -0
- package/dist/chunk-APV6MK5E.js +137 -0
- package/dist/chunk-ASEP3M2W.js +968 -0
- package/dist/chunk-AT4TNPWW.js +2931 -0
- package/dist/chunk-ATVKCSGT.js +250 -0
- package/dist/chunk-AURFMVCD.js +899 -0
- package/dist/chunk-AX6GWE7V.js +47 -0
- package/dist/chunk-AY5EWJPF.js +385 -0
- package/dist/chunk-BAV2HJXS.js +194 -0
- package/dist/chunk-BBHRN366.js +89 -0
- package/dist/chunk-BBPSTTPA.js +1004 -0
- package/dist/chunk-BDATUDDJ.js +992 -0
- package/dist/chunk-BKE5PG5L.js +20414 -0
- package/dist/chunk-BLPHWMDK.js +797 -0
- package/dist/chunk-BSE4R6XD.js +322 -0
- package/dist/chunk-BXGMAU4A.js +396 -0
- package/dist/chunk-BXI4NXL3.js +507 -0
- package/dist/chunk-BZU2O43F.js +1125 -0
- package/dist/chunk-C3L6YQ7P.js +147 -0
- package/dist/chunk-C57KJOUQ.js +121 -0
- package/dist/chunk-CCV7BPSY.js +118 -0
- package/dist/chunk-CD7ISXA4.js +826 -0
- package/dist/chunk-CE5OZBYY.js +21 -0
- package/dist/chunk-CFBIG37O.js +154 -0
- package/dist/chunk-CJC6GZ46.js +457 -0
- package/dist/chunk-CKCDGXIV.js +408 -0
- package/dist/chunk-CKNX3TTP.js +10338 -0
- package/dist/chunk-CLKSETWE.js +10 -0
- package/dist/chunk-CP7H3HSP.js +1552 -0
- package/dist/chunk-CPLEWXNZ.js +66 -0
- package/dist/chunk-CQM3A35X.js +844 -0
- package/dist/chunk-CS6P76D3.js +103 -0
- package/dist/chunk-CTLUBNCW.js +622 -0
- package/dist/chunk-CUSRN6RJ.js +632 -0
- package/dist/chunk-D5EP5D2W.js +26 -0
- package/dist/chunk-D5RYINO7.js +412 -0
- package/dist/chunk-DBFS4HMH.js +751 -0
- package/dist/chunk-DC4ZYMXJ.js +115 -0
- package/dist/chunk-DCBZP442.js +9243 -0
- package/dist/chunk-DCYF7EKG.js +26 -0
- package/dist/chunk-DDMAFNUK.js +5351 -0
- package/dist/chunk-DF5SLDE4.js +966 -0
- package/dist/chunk-DJFW7SMX.js +348 -0
- package/dist/chunk-DQFLLBJW.js +121 -0
- package/dist/chunk-DSFM56OA.js +1203 -0
- package/dist/chunk-DTY4APYV.js +218 -0
- package/dist/chunk-DXH5JBRS.js +437 -0
- package/dist/chunk-E3NTPHEO.js +194 -0
- package/dist/chunk-E74LUIID.js +302 -0
- package/dist/chunk-EI57M5BI.js +194 -0
- package/dist/chunk-EJ4ZAWLV.js +305 -0
- package/dist/chunk-EJGCZGUF.js +1483 -0
- package/dist/chunk-EK2U3YEG.js +486 -0
- package/dist/chunk-ELMFYNV7.js +109 -0
- package/dist/chunk-ENRGKNR5.js +25 -0
- package/dist/chunk-ETYDXCK7.js +679 -0
- package/dist/chunk-EYRH3W74.js +239 -0
- package/dist/chunk-F23DIWNN.js +1024 -0
- package/dist/chunk-F5JO7HAS.js +94 -0
- package/dist/chunk-F7ER6ANO.js +174 -0
- package/dist/chunk-FATUVIIN.js +374 -0
- package/dist/chunk-FC23CTXV.js +322 -0
- package/dist/chunk-FD4ERYEG.js +22 -0
- package/dist/chunk-FH7WOJAE.js +79 -0
- package/dist/chunk-FHOWVDF5.js +1342 -0
- package/dist/chunk-FKAUPHMR.js +96 -0
- package/dist/chunk-FONNFJXA.js +137 -0
- package/dist/chunk-FQ7FSZRW.js +75 -0
- package/dist/chunk-FRG57FPM.js +240 -0
- package/dist/chunk-FUVP64ER.js +31 -0
- package/dist/chunk-FYOILAN5.js +6818 -0
- package/dist/chunk-G35UYQ6Z.js +69 -0
- package/dist/chunk-G36W2T2S.js +1143 -0
- package/dist/chunk-G3L2L5XZ.js +49 -0
- package/dist/chunk-G4THLFV3.js +148 -0
- package/dist/chunk-GAFJEYLK.js +29 -0
- package/dist/chunk-GFGYB2ZU.js +1689 -0
- package/dist/chunk-GG6KJYW6.js +85 -0
- package/dist/chunk-GHHFDU2U.js +55 -0
- package/dist/chunk-GHTZSQB7.js +57 -0
- package/dist/chunk-GK37ILCX.js +1231 -0
- package/dist/chunk-GKN3HMN4.js +2809 -0
- package/dist/chunk-GNVDBVJP.js +156 -0
- package/dist/chunk-GPEL3FBM.js +924 -0
- package/dist/chunk-GPSNGPZS.js +1604 -0
- package/dist/chunk-GSGD5E4C.js +30 -0
- package/dist/chunk-GSUQJJXQ.js +331 -0
- package/dist/chunk-GSV226X6.js +110 -0
- package/dist/chunk-H2AYQIHW.js +224 -0
- package/dist/chunk-H66DXFS5.js +1374 -0
- package/dist/chunk-H6G7QRMQ.js +35 -0
- package/dist/chunk-H7JAGHZT.js +59 -0
- package/dist/chunk-H7ZEFECA.js +316 -0
- package/dist/chunk-HAC6EHZZ.js +79 -0
- package/dist/chunk-HIO33P3G.js +19 -0
- package/dist/chunk-HMVQIPH3.js +547 -0
- package/dist/chunk-HOYBYAOA.js +257 -0
- package/dist/chunk-HX5P3CSP.js +1784 -0
- package/dist/chunk-I46EG2XQ.js +65 -0
- package/dist/chunk-I4NPDSCB.js +279 -0
- package/dist/chunk-I5WRT7ZR.js +514 -0
- package/dist/chunk-IC7I6YIJ.js +301 -0
- package/dist/chunk-IDCSOZVN.js +78 -0
- package/dist/chunk-IFCBYECK.js +112 -0
- package/dist/chunk-IFZCPEY2.js +66 -0
- package/dist/chunk-IKFP2KUR.js +48 -0
- package/dist/chunk-IRFBN336.js +878 -0
- package/dist/chunk-IUXNGKA6.js +152 -0
- package/dist/chunk-IV2RVAN7.js +1356 -0
- package/dist/chunk-IZYIQQ7X.js +53 -0
- package/dist/chunk-J3UUR7DE.js +210 -0
- package/dist/chunk-J54VUAY2.js +566 -0
- package/dist/chunk-J62KPGII.js +464 -0
- package/dist/chunk-J6P3FWWW.js +130 -0
- package/dist/chunk-JKBHJQMT.js +434 -0
- package/dist/chunk-JLS65NUQ.js +14 -0
- package/dist/chunk-JOSYEJKY.js +141 -0
- package/dist/chunk-JPKSXKNB.js +94 -0
- package/dist/chunk-JQNJB2BB.js +2440 -0
- package/dist/chunk-JRVNSWSQ.js +2626 -0
- package/dist/chunk-JSWMRQYG.js +28 -0
- package/dist/chunk-JZKNBBIB.js +152 -0
- package/dist/chunk-K54Y7MLK.js +160 -0
- package/dist/chunk-K65546RE.js +183 -0
- package/dist/chunk-KM3F7SYE.js +702 -0
- package/dist/chunk-KNDYMVWN.js +102 -0
- package/dist/chunk-KNT72VR6.js +132 -0
- package/dist/chunk-KP6YILMU.js +482 -0
- package/dist/chunk-KQEBSK3T.js +274 -0
- package/dist/chunk-KT6VHSPF.js +84 -0
- package/dist/chunk-KTMW65UX.js +14 -0
- package/dist/chunk-KZ5CJVBP.js +1813 -0
- package/dist/chunk-L4427KOE.js +26 -0
- package/dist/chunk-L6OM4A22.js +189 -0
- package/dist/chunk-L7E6XCRA.js +1042 -0
- package/dist/chunk-L7JZLXQL.js +173 -0
- package/dist/chunk-LAMKBTXX.js +40 -0
- package/dist/chunk-LDCF2MKN.js +132 -0
- package/dist/chunk-LEIIWPWQ.js +78 -0
- package/dist/chunk-LFTQVYBB.js +31 -0
- package/dist/chunk-LG3PDRTR.js +2872 -0
- package/dist/chunk-LGHII5ET.js +23 -0
- package/dist/chunk-LH5QY22X.js +1205 -0
- package/dist/chunk-LIVJ6266.js +438 -0
- package/dist/chunk-LKRAR3IH.js +276 -0
- package/dist/chunk-LO23NABI.js +3063 -0
- package/dist/chunk-LO3X5NLS.js +326 -0
- package/dist/chunk-LPITVM7M.js +339 -0
- package/dist/chunk-LR5IJIGV.js +810 -0
- package/dist/chunk-LRGMHROI.js +79 -0
- package/dist/chunk-LUVKFFWR.js +315 -0
- package/dist/chunk-LWCBNGH6.js +35 -0
- package/dist/chunk-LWTN3WRV.js +260 -0
- package/dist/chunk-M3HQYKQX.js +308 -0
- package/dist/chunk-M6GVSPEH.js +1016 -0
- package/dist/chunk-MF5HP6XV.js +221 -0
- package/dist/chunk-MHLCBZJB.js +328 -0
- package/dist/chunk-MN6PL4AN.js +120 -0
- package/dist/chunk-MOEY65Z6.js +5678 -0
- package/dist/chunk-MQ4WA34C.js +206 -0
- package/dist/chunk-MWHF5V7U.js +223 -0
- package/dist/chunk-MWHLBPNU.js +211 -0
- package/dist/chunk-MWK5UZQ3.js +1068 -0
- package/dist/chunk-MYLH2S4G.js +64 -0
- package/dist/chunk-MZ5WYKNA.js +28 -0
- package/dist/chunk-N4VTRA7K.js +131 -0
- package/dist/chunk-N5DX4JAK.js +12 -0
- package/dist/chunk-N65O4XTH.js +5011 -0
- package/dist/chunk-NEQUQHOX.js +223 -0
- package/dist/chunk-NGK6PEOG.js +25 -0
- package/dist/chunk-NJVD4PWE.js +1033 -0
- package/dist/chunk-NM6PMYY4.js +768 -0
- package/dist/chunk-NN4PJDHP.js +708 -0
- package/dist/chunk-NP2V3L7K.js +226 -0
- package/dist/chunk-NRLY536M.js +260 -0
- package/dist/chunk-NRNE225N.js +1270 -0
- package/dist/chunk-NS2K2GTP.js +605 -0
- package/dist/chunk-NS33V3IM.js +51 -0
- package/dist/chunk-NSHASMZZ.js +71 -0
- package/dist/chunk-NWOYOA6H.js +378 -0
- package/dist/chunk-NWRUSEXG.js +322 -0
- package/dist/chunk-NZGDAWIP.js +1905 -0
- package/dist/chunk-O47UOKU3.js +395 -0
- package/dist/chunk-O4G2W3ST.js +9831 -0
- package/dist/chunk-O6V4PNLO.js +63 -0
- package/dist/chunk-OA4GPILI.js +624 -0
- package/dist/chunk-OGMSYL6J.js +227 -0
- package/dist/chunk-OIS7J26S.js +319 -0
- package/dist/chunk-OMX5VL43.js +79 -0
- package/dist/chunk-ON3G73BU.js +264 -0
- package/dist/chunk-OPYFYTWQ.js +220 -0
- package/dist/chunk-OQ6K46CK.js +1734 -0
- package/dist/chunk-OYZYO7TV.js +47 -0
- package/dist/chunk-OZ4REHIN.js +4386 -0
- package/dist/chunk-P47VN5F4.js +282 -0
- package/dist/chunk-P5I4X7A7.js +353 -0
- package/dist/chunk-P6VQROCO.js +594 -0
- package/dist/chunk-P7CQGPLS.js +30 -0
- package/dist/chunk-PCAGC5MR.js +1763 -0
- package/dist/chunk-PFTQBXS6.js +5181 -0
- package/dist/chunk-PGSPX4SU.js +15 -0
- package/dist/chunk-PIKLF7BM.js +432 -0
- package/dist/chunk-PJWJ3SAV.js +22 -0
- package/dist/chunk-PKIFMV72.js +100 -0
- package/dist/chunk-PL37JDNA.js +189 -0
- package/dist/chunk-PTH5E5XO.js +195 -0
- package/dist/chunk-PW6AXLQC.js +4128 -0
- package/dist/chunk-PYITM4N2.js +79 -0
- package/dist/chunk-PZ5AY32C.js +10 -0
- package/dist/chunk-Q3RO7N35.js +45 -0
- package/dist/chunk-Q3WIC6GQ.js +2670 -0
- package/dist/chunk-Q5XIXSHX.js +271 -0
- package/dist/chunk-QGJDMEOV.js +34 -0
- package/dist/chunk-QNOB37UH.js +58 -0
- package/dist/chunk-QOCOVLUH.js +462 -0
- package/dist/chunk-QTCJ5CC5.js +172 -0
- package/dist/chunk-QVFV5IP2.js +232 -0
- package/dist/chunk-QVZMFDYG.js +29 -0
- package/dist/chunk-R35PEUKH.js +511 -0
- package/dist/chunk-R4UWBC35.js +200 -0
- package/dist/chunk-R6MLCREW.js +44 -0
- package/dist/chunk-R6W23LKN.js +254 -0
- package/dist/chunk-R7XDDX2A.js +4007 -0
- package/dist/chunk-RDFRCT64.js +168 -0
- package/dist/chunk-RE5VZDFQ.js +58 -0
- package/dist/chunk-RGROYC2G.js +638 -0
- package/dist/chunk-RGWHUI5E.js +68 -0
- package/dist/chunk-RHVG6UNL.js +205 -0
- package/dist/chunk-RI4EX2QE.js +18 -0
- package/dist/chunk-RJAAKDRX.js +17 -0
- package/dist/chunk-ROMHOSMQ.js +185 -0
- package/dist/chunk-RWREC3KJ.js +517 -0
- package/dist/chunk-RYNETRDJ.js +118 -0
- package/dist/chunk-RZY7RKM5.js +500 -0
- package/dist/chunk-S2MNWFAG.js +52 -0
- package/dist/chunk-S2VQCZO4.js +22 -0
- package/dist/chunk-S54MKU6V.js +117 -0
- package/dist/chunk-S5OIDFRM.js +302 -0
- package/dist/chunk-SCW4ZF6R.js +150 -0
- package/dist/chunk-SLS2N4PH.js +2213 -0
- package/dist/chunk-SNLBO4BF.js +247 -0
- package/dist/chunk-SP3LFU6B.js +60 -0
- package/dist/chunk-SQQTDDQA.js +537 -0
- package/dist/chunk-STV6LYYE.js +45 -0
- package/dist/chunk-SURZ2WFE.js +326 -0
- package/dist/chunk-SZ7DL357.js +1047 -0
- package/dist/chunk-T5E4NQCM.js +5742 -0
- package/dist/chunk-TAJPCXSB.js +13 -0
- package/dist/chunk-TAOOYK3P.js +358 -0
- package/dist/chunk-TDSEX5CF.js +24 -0
- package/dist/chunk-TFTGP7Z7.js +174 -0
- package/dist/chunk-TGON4N4O.js +197 -0
- package/dist/chunk-TGQZHDY2.js +2373 -0
- package/dist/chunk-TJJZ2OKV.js +6783 -0
- package/dist/chunk-TRSVUPCX.js +25 -0
- package/dist/chunk-TSXGRLPR.js +107 -0
- package/dist/chunk-U2S3YEX3.js +224 -0
- package/dist/chunk-U45RTRGY.js +818 -0
- package/dist/chunk-U6C3VVWS.js +13 -0
- package/dist/chunk-UDMS5AD3.js +307 -0
- package/dist/chunk-UGX7GI37.js +94 -0
- package/dist/chunk-UH3JJSHK.js +1 -0
- package/dist/chunk-UHRBTZYY.js +208 -0
- package/dist/chunk-UHT2KRG5.js +193 -0
- package/dist/chunk-UI2F6DJ5.js +118 -0
- package/dist/chunk-UMAKSB4Z.js +284 -0
- package/dist/chunk-UR2IBGIM.js +521 -0
- package/dist/chunk-UUH2RVKS.js +585 -0
- package/dist/chunk-UVJ4PJDK.js +4533 -0
- package/dist/chunk-V53UG2OT.js +9924 -0
- package/dist/chunk-V72RMBM4.js +48 -0
- package/dist/chunk-VBNCKCCI.js +16 -0
- package/dist/chunk-VNSYKC25.js +24 -0
- package/dist/chunk-VPHVFK4A.js +1984 -0
- package/dist/chunk-VTLNJQ44.js +133 -0
- package/dist/chunk-VTW7HP3R.js +100 -0
- package/dist/chunk-VVGEPBPS.js +218 -0
- package/dist/chunk-VVLLL7I4.js +166 -0
- package/dist/chunk-VX3UMWB7.js +483 -0
- package/dist/chunk-VXF4Q3FW.js +20 -0
- package/dist/chunk-W3JMT2YY.js +712 -0
- package/dist/chunk-WDUORIHF.js +877 -0
- package/dist/chunk-WDVTY4U6.js +148 -0
- package/dist/chunk-WIFZCHEG.js +12566 -0
- package/dist/chunk-WIRN3F75.js +736 -0
- package/dist/chunk-WMPU2UOV.js +82 -0
- package/dist/chunk-WTOKKE2V.js +784 -0
- package/dist/chunk-WX5IRLF6.js +740 -0
- package/dist/chunk-WZ6U3O7I.js +1581 -0
- package/dist/chunk-XAONGNST.js +57 -0
- package/dist/chunk-XB5YLCLB.js +13 -0
- package/dist/chunk-XEP5NQ5Y.js +688 -0
- package/dist/chunk-XGYF2QMZ.js +351 -0
- package/dist/chunk-XHEKQENB.js +255 -0
- package/dist/chunk-XNO4FM6X.js +1124 -0
- package/dist/chunk-XPHKYHO4.js +878 -0
- package/dist/chunk-XPK2I4NC.js +405 -0
- package/dist/chunk-XPU2XADL.js +2175 -0
- package/dist/chunk-XQD2B6BB.js +507 -0
- package/dist/chunk-XT2ZQVTC.js +558 -0
- package/dist/chunk-XTXJGGUV.js +351 -0
- package/dist/chunk-XUXVDPSZ.js +50 -0
- package/dist/chunk-XV7NZL4B.js +57 -0
- package/dist/chunk-XZE4XARH.js +563 -0
- package/dist/chunk-XZYR5EDO.js +152 -0
- package/dist/chunk-Y3J4JMTU.js +61 -0
- package/dist/chunk-Y55LZN5U.js +52 -0
- package/dist/chunk-Y5VN7DMP.js +112 -0
- package/dist/chunk-Y67VAHN4.js +459 -0
- package/dist/chunk-YAXSKEUH.js +3544 -0
- package/dist/chunk-YB3JD4E6.js +975 -0
- package/dist/chunk-YESZJFK6.js +220 -0
- package/dist/chunk-YGXKOBQQ.js +292 -0
- package/dist/chunk-YMX76IOS.js +31 -0
- package/dist/chunk-YOTXESEH.js +257 -0
- package/dist/chunk-YTYIEZQY.js +119 -0
- package/dist/chunk-YVORHQ2S.js +579 -0
- package/dist/chunk-Z2CUQDAR.js +3429 -0
- package/dist/chunk-Z2KYOFHS.js +7000 -0
- package/dist/chunk-Z6PC7QX6.js +161 -0
- package/dist/chunk-ZDTGECTN.js +235 -0
- package/dist/chunk-ZIPWI2LY.js +151 -0
- package/dist/chunk-ZSKKMTQJ.js +554 -0
- package/dist/chunk-ZTT5IVPY.js +1160 -0
- package/dist/chunk-ZX65UI5W.js +78 -0
- package/dist/chunk-ZYG7GGKO.js +218 -0
- package/dist/chunk-ZZZUL4OF.js +38 -0
- package/dist/cli-utils-IXL26KJT.js +40 -0
- package/dist/cli-validation-WLXWLUJL.js +111 -0
- package/dist/commit-utils-P7CRJF5N.js +202 -0
- package/dist/compete-config-resolver-R5ULXMYH.js +76 -0
- package/dist/compete-request-62JESVET.js +364 -0
- package/dist/compete-results-QEPG6FZX.js +208 -0
- package/dist/competition-outcome-tracker-R32OOZ43.js +38 -0
- package/dist/config-JCOKXOQT.js +166 -0
- package/dist/config-audit-EL3GHXS7.js +360 -0
- package/dist/config-explain-TNILDEVJ.js +12 -0
- package/dist/config-loader-VPFPFO4B.js +38 -0
- package/dist/config-manager-7T4E2BQO.js +67 -0
- package/dist/config-validator-KKOHOW7O.js +72 -0
- package/dist/cost-YC42XUF3.js +53 -0
- package/dist/coverage-orchestrator-CISNKJC2.js +772 -0
- package/dist/coverage-scanner-YVD62MLH.js +12 -0
- package/dist/createCompetitionRunner-LJG6ZQSO.js +46 -0
- package/dist/data-migration-state-VJK64DLT.js +31 -0
- package/dist/db-migrator-XSEIOGWY.js +34 -0
- package/dist/discovery-JGYGDD25.js +25 -0
- package/dist/doctor-GJOYE7AE.js +186 -0
- package/dist/domain-analyzer-AWF3RP6E.js +29 -0
- package/dist/ensure-initialized-744DIZEZ.js +52 -0
- package/dist/entry.js +940 -0
- package/dist/environment-YS45U5ZI.js +81 -0
- package/dist/error-parser-core-PPZ36A4S.js +50 -0
- package/dist/escalation-ladder-MKTCEGSA.js +130 -0
- package/dist/evaluation-context-KV4GZNN7.js +35 -0
- package/dist/evidence-plan-resolver-GBSYSAPX.js +85 -0
- package/dist/execution-data-writer-3ODTBDLH.js +53 -0
- package/dist/execution-policy-applier-ZHD3DCF3.js +31 -0
- package/dist/execution-preflight-Z4Y64V3I.js +209 -0
- package/dist/execution-roles-R3DPPMD3.js +66 -0
- package/dist/exit-code-error-Z4SW2DVK.js +14 -0
- package/dist/external-temp-cleanup-N2RZH4QP.js +311 -0
- package/dist/factory-XBN7DFIM.js +149 -0
- package/dist/file-utils-XIVR2ZMA.js +47 -0
- package/dist/flywheel-autoloop-executor-UILGOFFK.js +249 -0
- package/dist/flywheel-competition-executor-JOBEXTRB.js +259 -0
- package/dist/flywheel-executor-6JFLA6J5.js +369 -0
- package/dist/flywheel-git-isolation-UC5NIEVU.js +48 -0
- package/dist/flywheel-manifest-5UC4VBMA.js +108 -0
- package/dist/flywheel-model-defaults-7CVKP53V.js +27 -0
- package/dist/flywheel-preflight-XEY5SHM6.js +104 -0
- package/dist/flywheel-resume-7HELH7QS.js +23 -0
- package/dist/flywheel-safety-7U7DPOAP.js +28 -0
- package/dist/flywheel-scope-decision-ZHKF6ZIN.js +11 -0
- package/dist/get-telemetry-logger-DPVWWIYY.js +73 -0
- package/dist/git-worktree-IY6V6DUO.js +18 -0
- package/dist/github-pr-manager-4LXEKJTA.js +211 -0
- package/dist/guidance-profile-config-E7XJL2VP.js +117 -0
- package/dist/guidance-profile-renderer-QK2YMAAP.js +21 -0
- package/dist/index-query-4JEUKA44.js +31 -0
- package/dist/index.js +108661 -0
- package/dist/instructions-6BURLQJC.js +64 -0
- package/dist/invoke-provider-auth-GRUYP44M.js +223 -0
- package/dist/judge-scoring.json +80 -0
- package/dist/judge-test.js +555 -0
- package/dist/lab-mode-OTWVHN63.js +39 -0
- package/dist/lab-utils-BJRFLGK4.js +23 -0
- package/dist/linux-host-class-UY4KVLFD.js +29 -0
- package/dist/llm-judge-executor-F7Z3HO77.js +133 -0
- package/dist/local-executor-CEHWE2PL.js +223 -0
- package/dist/logger-33TR6EE5.js +72 -0
- package/dist/loop-spec-XTLQAVZ7.js +191 -0
- package/dist/manager-2HJG5CLT.js +25 -0
- package/dist/matrix-prompt-builder-6UDN5X3F.js +497 -0
- package/dist/matrix-result-parser-7W35EUOV.js +15 -0
- package/dist/metadata-HQ43NDEK.js +22 -0
- package/dist/model-invocations-db-QBO4Y7WZ.js +38 -0
- package/dist/monitor-XGYBG2DW.js +72 -0
- package/dist/next-PB5ZXZKB.js +90 -0
- package/dist/next-ZBDK725T.js +69 -0
- package/dist/optimization-config-resolver-O6WKR47U.js +75 -0
- package/dist/package-3YCWIQQ5.js +10 -0
- package/dist/parallel-orchestrator-NL7YJJU2.js +297 -0
- package/dist/parallel-worker.js +279 -0
- package/dist/parse-git-status-PIAF3OTP.js +10 -0
- package/dist/path-security-ETTH6TZL.js +39 -0
- package/dist/pattern-bank-UNJR5ADV.js +15 -0
- package/dist/plan-backlog-list-fast-VAGYBV3J.js +270 -0
- package/dist/plan-backlog-show-fast-SG2CZPW3.js +406 -0
- package/dist/plan-sprint-list-fast-LURAX47V.js +162 -0
- package/dist/plan-sprint-status-fast-FJAODCHD.js +208 -0
- package/dist/plan-task-status-fast-HNYBMQI4.js +349 -0
- package/dist/plan-to-flywheel-Z4U3GWPN.js +456 -0
- package/dist/planning-backlog-3CQ5QQBQ.js +319 -0
- package/dist/poetic-root-V5DNXEAE.js +17 -0
- package/dist/preflight-OPANWMME.js +12 -0
- package/dist/process-registry-PDZJPTAS.js +27 -0
- package/dist/process-scanner-C22SYXYD.js +22 -0
- package/dist/processes-KMQUD7HY.js +29 -0
- package/dist/processes-json-fast-IBDXZ6WW.js +135 -0
- package/dist/profile-SE5EVXAP.js +149 -0
- package/dist/prompts-WZZPGWBX.js +168 -0
- package/dist/protected-branches-RNTEX7ET.js +57 -0
- package/dist/provider-aware-resource-manager-QBTZJMNP.js +18 -0
- package/dist/provider-matrix-N5X57V3Y.js +154 -0
- package/dist/provider-matrix-renderer-MZDWKKID.js +139 -0
- package/dist/provider-output-forwarding-DERID3TP.js +30 -0
- package/dist/provider-registry-J5NRBJKI.js +128 -0
- package/dist/provider-temp-cleanup-KYGLF6YU.js +14 -0
- package/dist/prune-engine-Q6E2DXBN.js +451 -0
- package/dist/quality-gate-YRHMQAEI.js +70 -0
- package/dist/quickstart-FIRLKGJI.js +62 -0
- package/dist/readiness-VZTMSG6J.js +96 -0
- package/dist/registry-YODFY77A.js +153 -0
- package/dist/repair-helpers-TTV5FEF4.js +117 -0
- package/dist/repo-root-2IGD4H7X.js +22 -0
- package/dist/resolution-engine-D6TX6FRG.js +127 -0
- package/dist/resolve-provider-cli-O2P6XZBZ.js +35 -0
- package/dist/restore-telemetry-db-BQL6DTQF.js +360 -0
- package/dist/result-streamer-WEHW476R.js +19 -0
- package/dist/routing-events-DHPUZDDF.js +27 -0
- package/dist/routing-history-store-UWCEXAP3.js +46 -0
- package/dist/routing-planner-F5WQEUOM.js +160 -0
- package/dist/run-poetic-tui-J6LWRJUP.js +13517 -0
- package/dist/run-simulation.js +376 -0
- package/dist/runner-CS2TNGHA.js +1092 -0
- package/dist/runner-HHH7ZN2Z.js +1571 -0
- package/dist/safety-OCC4KJJR.js +64 -0
- package/dist/safety-stanza-7SV3HGG3.js +60 -0
- package/dist/sandbox-32QXJEMV.js +81 -0
- package/dist/sandbox-RSO7VP6V.js +94 -0
- package/dist/schema-extensions.sql +234 -0
- package/dist/security-2KI5XNDG.js +17 -0
- package/dist/setup-bb-plugin-QAQWZCV5.js +738 -0
- package/dist/setup-claude-code-P7BTTSPO.js +317 -0
- package/dist/setup-temp-cleanup-GASCWZ2I.js +20 -0
- package/dist/shared-utils-D6U6UREP.js +37 -0
- package/dist/simple-artifact-goal-OTXNGZY6.js +25 -0
- package/dist/sprint-UIXPYOXQ.js +338 -0
- package/dist/sprint-bridge-4BHMSC3M.js +68 -0
- package/dist/sprint-execution-service-2B7WHL5L.js +311 -0
- package/dist/sprint-manager-IDQUDNQI.js +91 -0
- package/dist/sqlite-wrapper-UDWCTNLJ.js +17 -0
- package/dist/stale-cleanup-orchestrator-XWXV7KCZ.js +62 -0
- package/dist/storage-access-F6ALDN4S.js +22 -0
- package/dist/synthesis-XD6EDVHF.js +185 -0
- package/dist/task-classifier-LNLXSHOM.js +494 -0
- package/dist/task-list-fast-PTMA4LUC.js +147 -0
- package/dist/task-manager-XAT7GOWB.js +93 -0
- package/dist/task-splitter-U7DJ7W6A.js +79 -0
- package/dist/task-type-detector-OLVOMXZO.js +21 -0
- package/dist/task-type-resolver-KKNQYSB4.js +42 -0
- package/dist/telemetry-5HWBBXLX.js +131 -0
- package/dist/telemetry-health-tracker-DMSSGBZ2.js +11 -0
- package/dist/telemetry-path-resolver-7ICSG276.js +28 -0
- package/dist/telemetry-reliability-XUOZNZZ6.js +125 -0
- package/dist/token-estimator-MEQUL3EJ.js +37 -0
- package/dist/transports-KNSSV2V4.js +255 -0
- package/dist/unified-cost-tracker-4TFRBVPR.js +35 -0
- package/dist/unified-state-cleanup-LJPF2BOW.js +276 -0
- package/dist/universal-cost-calculator-BAJV35GA.js +42 -0
- package/dist/user-agents-GHUGCSKN.js +22 -0
- package/dist/variant-delivery-state-KLRNG6XA.js +33 -0
- package/dist/variant-state-cleanup-MZGGLQAJ.js +280 -0
- package/dist/variant-success-PJXWE3ZV.js +12 -0
- package/dist/variant-worker-KUWKBLAH.js +4196 -0
- package/dist/variant-worker.js +24 -0
- package/dist/variants-JE2FSAQS.js +233 -0
- package/dist/verification-toolchains-V24UHW6Q.js +28 -0
- package/dist/verify-commands-H3DTCSGI.js +37 -0
- package/dist/warning-EGQGUUIO.js +12 -0
- package/dist/work-edge-RWEDWJZF.js +49 -0
- package/dist/work-item-writer-WHAVZXAY.js +19 -0
- package/dist/worker-7SA2VS2Y.js +1435 -0
- package/dist/worker.js +19 -0
- package/dist/worktree-metrics-3DFUYVYE.js +26 -0
- package/dist/writeback-VZKHYHSN.js +44 -0
- package/docs/CLI_REFERENCE.md +7042 -0
- package/docs/PROVIDER_SETUP.md +1570 -0
- package/docs/README.md +96 -0
- package/docs/TROUBLESHOOTING.md +2232 -0
- package/docs/getting-started/QUICK_START.md +294 -0
- package/docs/getting-started/README.md +188 -0
- package/docs/getting-started/SETUP.md +52 -0
- package/docs/reference/AGENT_CONTEXT.md +38 -0
- package/docs/reference/PROVIDER_RELEASE_TIERS.md +34 -0
- package/docs/reference/README.md +381 -0
- package/docs/reference/SECURITY.md +262 -0
- package/integrations/bb-plugin-poetic/README.md +362 -0
- package/integrations/bb-plugin-poetic/app.css +341 -0
- package/integrations/bb-plugin-poetic/app.tsx +5059 -0
- package/integrations/bb-plugin-poetic/package.json +45 -0
- package/integrations/bb-plugin-poetic/server.ts +444 -0
- package/integrations/bb-plugin-poetic/src/adapter.ts +3312 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-model.ts +150 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-schema.ts +355 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-service.ts +419 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-view.ts +1026 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-schema.ts +122 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-service.ts +105 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-view.ts +138 -0
- package/integrations/bb-plugin-poetic/src/contract.ts +615 -0
- package/integrations/bb-plugin-poetic/src/finalize-model.ts +205 -0
- package/integrations/bb-plugin-poetic/src/finalize-schema.ts +158 -0
- package/integrations/bb-plugin-poetic/src/finalize-service.ts +382 -0
- package/integrations/bb-plugin-poetic/src/finalize-view.ts +159 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-readback.ts +303 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-schema.ts +33 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-view.ts +459 -0
- package/integrations/bb-plugin-poetic/src/model.ts +1302 -0
- package/integrations/bb-plugin-poetic/src/monitor-service.ts +233 -0
- package/integrations/bb-plugin-poetic/src/panel-read-ux.ts +97 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-schema.ts +166 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-service.ts +480 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-view.ts +109 -0
- package/integrations/bb-plugin-poetic/src/planning-readback.ts +204 -0
- package/integrations/bb-plugin-poetic/src/planning-service.ts +594 -0
- package/integrations/bb-plugin-poetic/src/planning-workspace-schema.ts +95 -0
- package/integrations/bb-plugin-poetic/src/planning-workspace.ts +323 -0
- package/integrations/bb-plugin-poetic/src/poll-handoff.ts +136 -0
- package/integrations/bb-plugin-poetic/src/project-target-server.ts +21 -0
- package/integrations/bb-plugin-poetic/src/project-target.ts +45 -0
- package/integrations/bb-plugin-poetic/src/reference-index.ts +192 -0
- package/integrations/bb-plugin-poetic/src/repository-schema.ts +30 -0
- package/integrations/bb-plugin-poetic/src/result-explorer-view.ts +597 -0
- package/integrations/bb-plugin-poetic/src/result-readback-model.ts +163 -0
- package/integrations/bb-plugin-poetic/src/result-readback-schema.ts +319 -0
- package/integrations/bb-plugin-poetic/src/result-readback-service.ts +333 -0
- package/integrations/bb-plugin-poetic/src/run-page-view.ts +177 -0
- package/integrations/bb-plugin-poetic/src/setup-config-schema.ts +291 -0
- package/integrations/bb-plugin-poetic/src/setup-config-service.ts +590 -0
- package/integrations/bb-plugin-poetic/src/setup-config-view.ts +413 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-model.ts +150 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-schema.ts +136 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-service.ts +341 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-view.ts +116 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-model.ts +714 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-readback.ts +231 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-schema.ts +601 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-service.ts +775 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-view.ts +1221 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-model.ts +567 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-schema.ts +635 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-service.ts +842 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-view.ts +1227 -0
- package/integrations/bb-plugin-poetic/src/task-detail-readback.ts +56 -0
- package/integrations/bb-plugin-poetic/src/task-detail-schema.ts +144 -0
- package/integrations/bb-plugin-poetic/src/task-navigation.ts +446 -0
- package/integrations/bb-plugin-poetic/src/workbench-route.ts +74 -0
- package/integrations/bb-plugin-poetic/tests/host-contract.test.ts +773 -0
- package/integrations/bb-plugin-poetic/tsconfig.json +18 -0
- package/integrations/bb-plugin-poetic/types/PROVENANCE.json +52 -0
- package/integrations/bb-plugin-poetic/types/bb-plugin-sdk-app.d.ts +1444 -0
- package/integrations/bb-plugin-poetic/types/bb-plugin-sdk.d.ts +13030 -0
- package/integrations/bb-plugin-poetic/vitest.host.config.ts +71 -0
- package/npm-shrinkwrap.json +4950 -0
- package/package.json +331 -0
- package/schemas/README.md +80 -0
- package/schemas/config-v1.schema.json +136 -0
- package/schemas/execution-config.schema.json +47 -0
- package/schemas/judge-scoring.schema.json +370 -0
- package/schemas/poetic.config.schema.json +2214 -0
- package/schemas/provider-config.schema.json +203 -0
- package/schemas/telemetry-config.schema.json +79 -0
- package/schemas/user-preferences.schema.json +141 -0
- package/scripts/assert-node-runtime.mjs +140 -0
- package/scripts/preflight-native.mjs +49 -0
- package/scripts/preinstall-node-check.mjs +78 -0
- package/scripts/setup-git-hooks.mjs +24 -0
- package/scripts/sync.sh +2722 -0
- package/scripts/write-node-launcher.sh +108 -0
- package/src/resources/gemini/slash-packs/default/plan.toml +15 -0
- package/src/resources/gemini/slash-packs/default/summary.toml +16 -0
- package/src/resources/gemini/slash-packs/default/tests.toml +16 -0
- package/templates/.poetic/README.md +37 -0
- package/templates/.poetic/agents/README.md +296 -0
- package/templates/.poetic/agents/api-documenter.md +147 -0
- package/templates/.poetic/agents/backend-architect.md +31 -0
- package/templates/.poetic/agents/code-reviewer.md +157 -0
- package/templates/.poetic/agents/data-scientist.md +179 -0
- package/templates/.poetic/agents/database-optimizer.md +145 -0
- package/templates/.poetic/agents/debugger.md +31 -0
- package/templates/.poetic/agents/deployment-engineer.md +164 -0
- package/templates/.poetic/agents/devops-troubleshooter.md +139 -0
- package/templates/.poetic/agents/frontend-developer.md +150 -0
- package/templates/.poetic/agents/javascript-pro.md +36 -0
- package/templates/.poetic/agents/performance-engineer.md +151 -0
- package/templates/.poetic/agents/python-pro.md +137 -0
- package/templates/.poetic/agents/test-automator.md +147 -0
- package/templates/.poetic/agents/typescript-pro.md +34 -0
- package/templates/.poetic/config/poetic.config.jsonc +69 -0
- package/templates/.poetic/config/project-context.template.json +6 -0
- package/templates/.poetic/config/task-type-aliases.presets/kanban.yaml +14 -0
- package/templates/.poetic/config/task-type-aliases.presets/scrum.yaml +17 -0
- package/templates/.poetic/config/task-type-aliases.presets/xp.yaml +12 -0
- package/templates/.poetic/config/task-type-aliases.yaml +28 -0
- package/templates/.poetic/gitignore.template +55 -0
- package/templates/.poetic/task-types/analysis.yaml +40 -0
- package/templates/.poetic/task-types/architecture.yaml +38 -0
- package/templates/.poetic/task-types/doc.yaml +35 -0
- package/templates/.poetic/task-types/feature.yaml +23 -0
- package/templates/.poetic/task-types/general.yaml +6 -0
- package/templates/.poetic/task-types/security.yaml +39 -0
- package/templates/AGENTS.template.md +99 -0
- package/templates/CLAUDE.template.md +1 -0
- package/templates/GEMINI.template.md +1 -0
- package/templates/builtin-workflows/code-review.yaml +73 -0
- package/templates/builtin-workflows/compete-streak.yaml +78 -0
- package/templates/builtin-workflows/hello-verify.yaml +10 -0
- package/templates/builtin-workflows/judge-regression.yaml +114 -0
- package/templates/builtin-workflows/skills/code-review/SKILL.md +60 -0
- package/templates/builtin-workflows/skills/hello-verify/SKILL.md +6 -0
- package/templates/guard-kit/GUARD_SETUP.md.template +255 -0
- package/templates/guard-kit/check.mjs.template +777 -0
- package/templates/guard-kit/config.json.template +6 -0
- package/templates/guard-kit/poetic-guard.yml.template +189 -0
- package/templates/profiles/README.md +56 -0
- package/templates/profiles/frontier-claude.json +27 -0
- package/templates/profiles/frontier-codex.json +26 -0
|
@@ -0,0 +1,3679 @@
|
|
|
1
|
+
import {
|
|
2
|
+
PromptManager,
|
|
3
|
+
VariantGenerator
|
|
4
|
+
} from "./chunk-Q3WIC6GQ.js";
|
|
5
|
+
import {
|
|
6
|
+
PromptStore
|
|
7
|
+
} from "./chunk-OQ6K46CK.js";
|
|
8
|
+
import {
|
|
9
|
+
BUILT_IN_VARIANT_STRATEGIES
|
|
10
|
+
} from "./chunk-VXF4Q3FW.js";
|
|
11
|
+
import {
|
|
12
|
+
DEFAULT_EVALUATION_CRITERIA,
|
|
13
|
+
getActiveCriteria
|
|
14
|
+
} from "./chunk-6Y5TWI7H.js";
|
|
15
|
+
import {
|
|
16
|
+
TelemetryLogger
|
|
17
|
+
} from "./chunk-CKNX3TTP.js";
|
|
18
|
+
import {
|
|
19
|
+
JUDGE_DEFAULTS
|
|
20
|
+
} from "./chunk-ETYDXCK7.js";
|
|
21
|
+
import {
|
|
22
|
+
parseJsonWithContext
|
|
23
|
+
} from "./chunk-QVZMFDYG.js";
|
|
24
|
+
import {
|
|
25
|
+
ProviderRegistry
|
|
26
|
+
} from "./chunk-AT4TNPWW.js";
|
|
27
|
+
import {
|
|
28
|
+
resolvePoeticDirFromContext
|
|
29
|
+
} from "./chunk-UI2F6DJ5.js";
|
|
30
|
+
import {
|
|
31
|
+
emitWarning,
|
|
32
|
+
isTelemetryDebug
|
|
33
|
+
} from "./chunk-EJGCZGUF.js";
|
|
34
|
+
|
|
35
|
+
// src/core/optimization/ActivationAuthority.ts
|
|
36
|
+
var TunerActivationAuthority = class {
|
|
37
|
+
constructor(tuner, telemetry) {
|
|
38
|
+
this.tuner = tuner;
|
|
39
|
+
this.telemetry = telemetry;
|
|
40
|
+
}
|
|
41
|
+
tuner;
|
|
42
|
+
telemetry;
|
|
43
|
+
async activate(request) {
|
|
44
|
+
const decision = {
|
|
45
|
+
shouldActivate: true,
|
|
46
|
+
candidateVersionId: request.candidateVersionId,
|
|
47
|
+
currentVersionId: request.sourceVersionId,
|
|
48
|
+
improvementPercentage: request.improvementPercentage,
|
|
49
|
+
statisticalSignificance: 1,
|
|
50
|
+
confidence: request.confidence,
|
|
51
|
+
safetyChecks: [],
|
|
52
|
+
recommendation: "activate",
|
|
53
|
+
rationale: request.reason,
|
|
54
|
+
decidedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
55
|
+
};
|
|
56
|
+
const record = await this.tuner.executeActivation(request.promptId, decision, {
|
|
57
|
+
skipSafetyChecks: request.skipSafetyChecks
|
|
58
|
+
});
|
|
59
|
+
const activated = record !== null;
|
|
60
|
+
if (activated && this.telemetry) {
|
|
61
|
+
this.telemetry.logCostEvent("prompt:activate:auto", {
|
|
62
|
+
promptId: request.promptId,
|
|
63
|
+
versionId: request.candidateVersionId,
|
|
64
|
+
improvementPercentage: request.improvementPercentage,
|
|
65
|
+
safetyChecksPassed: request.safetyChecksPassed ?? null
|
|
66
|
+
});
|
|
67
|
+
}
|
|
68
|
+
return {
|
|
69
|
+
activated,
|
|
70
|
+
record,
|
|
71
|
+
reason: request.reason
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
scheduleMonitoring(activationId) {
|
|
75
|
+
void this.tuner.monitorActivation(activationId).catch(() => {
|
|
76
|
+
});
|
|
77
|
+
}
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
// src/prompts/auto-tuner.ts
|
|
81
|
+
import { randomBytes } from "crypto";
|
|
82
|
+
|
|
83
|
+
// src/prompts/evaluator.ts
|
|
84
|
+
var PromptEvaluator = class {
|
|
85
|
+
store;
|
|
86
|
+
telemetry;
|
|
87
|
+
constructor(options = {}) {
|
|
88
|
+
this.store = new PromptStore(options);
|
|
89
|
+
this.telemetry = options.telemetryLogger || TelemetryLogger.getInstance(options);
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Evaluate a prompt version against criteria
|
|
93
|
+
*/
|
|
94
|
+
async evaluate(versionId, criteria = {}, variant) {
|
|
95
|
+
try {
|
|
96
|
+
const version = this.store.getVersion(versionId);
|
|
97
|
+
if (!version) {
|
|
98
|
+
throw new Error(`Version ${versionId} not found`);
|
|
99
|
+
}
|
|
100
|
+
const performance = await this.calculatePerformanceMetrics(version, variant);
|
|
101
|
+
const defaultCriteria = {
|
|
102
|
+
minSuccessRate: criteria.minSuccessRate ?? 0.7,
|
|
103
|
+
// 70%
|
|
104
|
+
minCompleteness: criteria.minCompleteness ?? 70,
|
|
105
|
+
maxAvgDuration: criteria.maxAvgDuration ?? 300,
|
|
106
|
+
// 5 minutes
|
|
107
|
+
minCodeDeliveryRate: criteria.minCodeDeliveryRate ?? 0.8,
|
|
108
|
+
// 80%
|
|
109
|
+
minSampleSize: criteria.minSampleSize ?? 5
|
|
110
|
+
};
|
|
111
|
+
const hasEnoughData = performance.sampleSize >= defaultCriteria.minSampleSize;
|
|
112
|
+
if (!hasEnoughData) {
|
|
113
|
+
return {
|
|
114
|
+
versionId,
|
|
115
|
+
promptId: version.promptId,
|
|
116
|
+
variant,
|
|
117
|
+
meetsStandards: false,
|
|
118
|
+
performance,
|
|
119
|
+
criteriaResults: [],
|
|
120
|
+
recommendation: "needs_data",
|
|
121
|
+
rationale: `Insufficient data for evaluation (${performance.sampleSize}/${defaultCriteria.minSampleSize} executions)`,
|
|
122
|
+
evaluatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
const criteriaResults = [
|
|
126
|
+
{
|
|
127
|
+
criterion: "Success Rate",
|
|
128
|
+
required: defaultCriteria.minSuccessRate,
|
|
129
|
+
actual: performance.successRate,
|
|
130
|
+
passed: performance.successRate >= defaultCriteria.minSuccessRate
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
criterion: "Completeness",
|
|
134
|
+
required: defaultCriteria.minCompleteness,
|
|
135
|
+
actual: performance.avgCompleteness,
|
|
136
|
+
passed: performance.avgCompleteness >= defaultCriteria.minCompleteness
|
|
137
|
+
},
|
|
138
|
+
{
|
|
139
|
+
criterion: "Duration",
|
|
140
|
+
required: defaultCriteria.maxAvgDuration,
|
|
141
|
+
actual: performance.avgDuration,
|
|
142
|
+
passed: performance.avgDuration <= defaultCriteria.maxAvgDuration
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
criterion: "Code Delivery Rate",
|
|
146
|
+
required: defaultCriteria.minCodeDeliveryRate,
|
|
147
|
+
actual: performance.avgCodeDeliveryRate,
|
|
148
|
+
passed: performance.avgCodeDeliveryRate >= defaultCriteria.minCodeDeliveryRate
|
|
149
|
+
}
|
|
150
|
+
];
|
|
151
|
+
const meetsStandards = criteriaResults.every((c) => c.passed);
|
|
152
|
+
const failedCriteria = criteriaResults.filter((c) => !c.passed);
|
|
153
|
+
let recommendation;
|
|
154
|
+
let rationale;
|
|
155
|
+
if (meetsStandards) {
|
|
156
|
+
recommendation = "keep";
|
|
157
|
+
rationale = "Prompt meets all quality standards and performs well";
|
|
158
|
+
} else if (failedCriteria.length === 1) {
|
|
159
|
+
recommendation = "improve";
|
|
160
|
+
const failed = failedCriteria[0];
|
|
161
|
+
if (!failed) {
|
|
162
|
+
rationale = "Prompt needs improvement";
|
|
163
|
+
} else {
|
|
164
|
+
rationale = `Prompt needs improvement in ${failed.criterion.toLowerCase()} (${failed.actual.toFixed(2)} vs ${failed.required.toFixed(2)} required)`;
|
|
165
|
+
}
|
|
166
|
+
} else if (performance.successRate < 0.5) {
|
|
167
|
+
recommendation = "replace";
|
|
168
|
+
rationale = `Prompt has low success rate (${(performance.successRate * 100).toFixed(1)}%) - consider replacing`;
|
|
169
|
+
} else {
|
|
170
|
+
recommendation = "improve";
|
|
171
|
+
rationale = `Prompt fails ${failedCriteria.length} criteria: ${failedCriteria.map((c) => c.criterion).join(", ")}`;
|
|
172
|
+
}
|
|
173
|
+
return {
|
|
174
|
+
versionId,
|
|
175
|
+
promptId: version.promptId,
|
|
176
|
+
variant,
|
|
177
|
+
meetsStandards,
|
|
178
|
+
performance,
|
|
179
|
+
criteriaResults,
|
|
180
|
+
recommendation,
|
|
181
|
+
rationale,
|
|
182
|
+
evaluatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
183
|
+
};
|
|
184
|
+
} catch (error) {
|
|
185
|
+
throw new Error(`Failed to evaluate version ${versionId}: ${error.message}`);
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
/**
|
|
189
|
+
* Analyze trends for a prompt version over time
|
|
190
|
+
*/
|
|
191
|
+
async analyzeTrends(versionId, variant, minDataPoints = 10) {
|
|
192
|
+
try {
|
|
193
|
+
const version = this.store.getVersion(versionId);
|
|
194
|
+
if (!version) {
|
|
195
|
+
throw new Error(`Version ${versionId} not found`);
|
|
196
|
+
}
|
|
197
|
+
const variantStr = variant || "conservative";
|
|
198
|
+
const executions = this.telemetry.getVariantPerformance(variantStr, {
|
|
199
|
+
limit: 100,
|
|
200
|
+
promptVersion: versionId
|
|
201
|
+
});
|
|
202
|
+
if (executions.length < minDataPoints) {
|
|
203
|
+
return {
|
|
204
|
+
versionId,
|
|
205
|
+
promptId: version.promptId,
|
|
206
|
+
variant,
|
|
207
|
+
dataPoints: executions.length,
|
|
208
|
+
trends: {
|
|
209
|
+
successRate: "insufficient_data",
|
|
210
|
+
completeness: "insufficient_data",
|
|
211
|
+
duration: "insufficient_data"
|
|
212
|
+
},
|
|
213
|
+
insights: [
|
|
214
|
+
`Insufficient data for trend analysis (${executions.length}/${minDataPoints} data points)`
|
|
215
|
+
],
|
|
216
|
+
analyzedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
const recentWindow = Math.floor(executions.length / 3);
|
|
220
|
+
const olderData = executions.slice(0, recentWindow);
|
|
221
|
+
const recentData = executions.slice(-recentWindow);
|
|
222
|
+
const recentSuccessRate = recentData.filter((e) => e.verdict === "success").length / recentData.length;
|
|
223
|
+
const olderSuccessRate = olderData.filter((e) => e.verdict === "success").length / olderData.length;
|
|
224
|
+
const successRateTrend = this.determineTrend(recentSuccessRate, olderSuccessRate);
|
|
225
|
+
const recentCompleteness = recentData.reduce((sum, e) => sum + (e.completeness || 0), 0) / recentData.length;
|
|
226
|
+
const olderCompleteness = olderData.reduce((sum, e) => sum + (e.completeness || 0), 0) / olderData.length;
|
|
227
|
+
const completenessTrend = this.determineTrend(recentCompleteness, olderCompleteness);
|
|
228
|
+
const recentDuration = recentData.reduce((sum, e) => sum + (e.total_duration_sec || 0), 0) / recentData.length;
|
|
229
|
+
const olderDuration = olderData.reduce((sum, e) => sum + (e.total_duration_sec || 0), 0) / olderData.length;
|
|
230
|
+
const durationTrend = this.determineTrend(olderDuration, recentDuration);
|
|
231
|
+
const insights = [];
|
|
232
|
+
if (successRateTrend === "improving") {
|
|
233
|
+
insights.push(
|
|
234
|
+
`Success rate is improving (${(olderSuccessRate * 100).toFixed(1)}% \u2192 ${(recentSuccessRate * 100).toFixed(1)}%)`
|
|
235
|
+
);
|
|
236
|
+
} else if (successRateTrend === "declining") {
|
|
237
|
+
insights.push(
|
|
238
|
+
`Success rate is declining (${(olderSuccessRate * 100).toFixed(1)}% \u2192 ${(recentSuccessRate * 100).toFixed(1)}%)`
|
|
239
|
+
);
|
|
240
|
+
}
|
|
241
|
+
if (completenessTrend === "improving") {
|
|
242
|
+
insights.push(
|
|
243
|
+
`Completeness is improving (${olderCompleteness.toFixed(1)} \u2192 ${recentCompleteness.toFixed(1)})`
|
|
244
|
+
);
|
|
245
|
+
} else if (completenessTrend === "declining") {
|
|
246
|
+
insights.push(
|
|
247
|
+
`Completeness is declining (${olderCompleteness.toFixed(1)} \u2192 ${recentCompleteness.toFixed(1)})`
|
|
248
|
+
);
|
|
249
|
+
}
|
|
250
|
+
if (durationTrend === "improving") {
|
|
251
|
+
insights.push(
|
|
252
|
+
`Execution time is improving (${olderDuration.toFixed(0)}s \u2192 ${recentDuration.toFixed(0)}s)`
|
|
253
|
+
);
|
|
254
|
+
} else if (durationTrend === "declining") {
|
|
255
|
+
insights.push(
|
|
256
|
+
`Execution time is declining (${olderDuration.toFixed(0)}s \u2192 ${recentDuration.toFixed(0)}s)`
|
|
257
|
+
);
|
|
258
|
+
}
|
|
259
|
+
if (insights.length === 0) {
|
|
260
|
+
insights.push("Performance is stable across all metrics");
|
|
261
|
+
}
|
|
262
|
+
return {
|
|
263
|
+
versionId,
|
|
264
|
+
promptId: version.promptId,
|
|
265
|
+
variant,
|
|
266
|
+
dataPoints: executions.length,
|
|
267
|
+
trends: {
|
|
268
|
+
successRate: successRateTrend,
|
|
269
|
+
completeness: completenessTrend,
|
|
270
|
+
duration: durationTrend
|
|
271
|
+
},
|
|
272
|
+
insights,
|
|
273
|
+
analyzedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
274
|
+
};
|
|
275
|
+
} catch (error) {
|
|
276
|
+
throw new Error(
|
|
277
|
+
`Failed to analyze trends for version ${versionId}: ${error.message}`
|
|
278
|
+
);
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
/**
|
|
282
|
+
* Calculate performance metrics from telemetry data
|
|
283
|
+
*/
|
|
284
|
+
async calculatePerformanceMetrics(version, variant) {
|
|
285
|
+
try {
|
|
286
|
+
const cachedMetrics = this.store.getPerformanceMetrics(version.id, variant);
|
|
287
|
+
const variantStr = variant || "conservative";
|
|
288
|
+
const executions = this.telemetry.getVariantPerformance(variantStr, {
|
|
289
|
+
limit: 1e3,
|
|
290
|
+
promptVersion: version.id
|
|
291
|
+
});
|
|
292
|
+
if (executions.length === 0) {
|
|
293
|
+
if (cachedMetrics) {
|
|
294
|
+
return {
|
|
295
|
+
...cachedMetrics,
|
|
296
|
+
calculatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
297
|
+
};
|
|
298
|
+
}
|
|
299
|
+
return {
|
|
300
|
+
promptId: version.promptId,
|
|
301
|
+
versionId: version.id,
|
|
302
|
+
variant,
|
|
303
|
+
totalExecutions: 0,
|
|
304
|
+
successfulExecutions: 0,
|
|
305
|
+
failedExecutions: 0,
|
|
306
|
+
successRate: 0,
|
|
307
|
+
avgCompleteness: 0,
|
|
308
|
+
avgCodeDeliveryRate: 0,
|
|
309
|
+
avgDuration: 0,
|
|
310
|
+
minCompleteness: 0,
|
|
311
|
+
maxCompleteness: 0,
|
|
312
|
+
stdDevCompleteness: 0,
|
|
313
|
+
sampleSize: 0,
|
|
314
|
+
firstExecutionAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
315
|
+
lastExecutionAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
316
|
+
calculatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
317
|
+
};
|
|
318
|
+
}
|
|
319
|
+
const successfulExecutions = executions.filter((e) => e.verdict === "success").length;
|
|
320
|
+
const failedExecutions = executions.length - successfulExecutions;
|
|
321
|
+
const successRate = successfulExecutions / executions.length;
|
|
322
|
+
const completenessValues = executions.map((e) => e.completeness || 0).filter((v) => v > 0);
|
|
323
|
+
const avgCompleteness = completenessValues.reduce((sum, v) => sum + v, 0) / completenessValues.length || 0;
|
|
324
|
+
const minCompleteness = completenessValues.length > 0 ? Math.min(...completenessValues) : 0;
|
|
325
|
+
const maxCompleteness = completenessValues.length > 0 ? Math.max(...completenessValues) : 0;
|
|
326
|
+
const variance = completenessValues.reduce((sum, v) => sum + (v - avgCompleteness) ** 2, 0) / completenessValues.length || 0;
|
|
327
|
+
const stdDevCompleteness = Math.sqrt(variance);
|
|
328
|
+
const codeDeliveredCount = executions.filter((e) => e.code_delivered).length;
|
|
329
|
+
const avgCodeDeliveryRate = codeDeliveredCount / executions.length;
|
|
330
|
+
const durationValues = executions.map((e) => e.total_duration_sec || 0).filter((v) => v > 0);
|
|
331
|
+
const avgDuration = durationValues.reduce((sum, v) => sum + v, 0) / durationValues.length || 0;
|
|
332
|
+
const timestamps = executions.map((e) => e.started_at).filter(Boolean).sort();
|
|
333
|
+
const firstExecutionAt = timestamps[0] || (/* @__PURE__ */ new Date()).toISOString();
|
|
334
|
+
const lastExecutionAt = timestamps[timestamps.length - 1] || (/* @__PURE__ */ new Date()).toISOString();
|
|
335
|
+
const metrics = {
|
|
336
|
+
promptId: version.promptId,
|
|
337
|
+
versionId: version.id,
|
|
338
|
+
variant,
|
|
339
|
+
totalExecutions: executions.length,
|
|
340
|
+
successfulExecutions,
|
|
341
|
+
failedExecutions,
|
|
342
|
+
successRate,
|
|
343
|
+
avgCompleteness,
|
|
344
|
+
avgCodeDeliveryRate,
|
|
345
|
+
avgDuration,
|
|
346
|
+
minCompleteness,
|
|
347
|
+
maxCompleteness,
|
|
348
|
+
stdDevCompleteness,
|
|
349
|
+
sampleSize: executions.length,
|
|
350
|
+
firstExecutionAt,
|
|
351
|
+
lastExecutionAt,
|
|
352
|
+
calculatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
353
|
+
};
|
|
354
|
+
await this.store.updatePerformanceMetrics(metrics);
|
|
355
|
+
return metrics;
|
|
356
|
+
} catch (error) {
|
|
357
|
+
throw new Error(`Failed to calculate performance metrics: ${error.message}`);
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
/**
|
|
361
|
+
* Batch evaluate all versions of a prompt
|
|
362
|
+
*/
|
|
363
|
+
async evaluateAllVersions(promptId, criteria) {
|
|
364
|
+
try {
|
|
365
|
+
const versions = this.store.listVersions(promptId);
|
|
366
|
+
const results = [];
|
|
367
|
+
for (const version of versions) {
|
|
368
|
+
try {
|
|
369
|
+
const result = await this.evaluate(version.id, criteria);
|
|
370
|
+
results.push(result);
|
|
371
|
+
} catch (error) {
|
|
372
|
+
emitWarning(
|
|
373
|
+
"VAL-001",
|
|
374
|
+
{
|
|
375
|
+
taskId: "prompt-evaluator",
|
|
376
|
+
subsystem: "prompts",
|
|
377
|
+
executionPhase: "validation"
|
|
378
|
+
},
|
|
379
|
+
{
|
|
380
|
+
severity: "ERROR" /* ERROR */,
|
|
381
|
+
metadata: {
|
|
382
|
+
error: error.message,
|
|
383
|
+
operation: "evaluateVersion",
|
|
384
|
+
versionId: version.id,
|
|
385
|
+
promptId
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
);
|
|
389
|
+
}
|
|
390
|
+
}
|
|
391
|
+
return results;
|
|
392
|
+
} catch (error) {
|
|
393
|
+
throw new Error(
|
|
394
|
+
`Failed to evaluate versions for prompt ${promptId}: ${error.message}`
|
|
395
|
+
);
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
/**
|
|
399
|
+
* Compare performance across variants for a single prompt version
|
|
400
|
+
*/
|
|
401
|
+
async compareVariantPerformance(versionId) {
|
|
402
|
+
try {
|
|
403
|
+
const version = this.store.getVersion(versionId);
|
|
404
|
+
if (!version) {
|
|
405
|
+
throw new Error(`Version ${versionId} not found`);
|
|
406
|
+
}
|
|
407
|
+
const variants = [...BUILT_IN_VARIANT_STRATEGIES];
|
|
408
|
+
const variantMetrics = [];
|
|
409
|
+
for (const variant of variants) {
|
|
410
|
+
try {
|
|
411
|
+
const metrics = await this.calculatePerformanceMetrics(version, variant);
|
|
412
|
+
if (metrics.sampleSize > 0) {
|
|
413
|
+
const evaluation = await this.evaluate(versionId, {}, variant);
|
|
414
|
+
variantMetrics.push({ variant, metrics, evaluation });
|
|
415
|
+
}
|
|
416
|
+
} catch (error) {
|
|
417
|
+
emitWarning(
|
|
418
|
+
"VAL-001",
|
|
419
|
+
{
|
|
420
|
+
taskId: "prompt-evaluator",
|
|
421
|
+
subsystem: "prompts",
|
|
422
|
+
executionPhase: "validation"
|
|
423
|
+
},
|
|
424
|
+
{
|
|
425
|
+
severity: "ERROR" /* ERROR */,
|
|
426
|
+
metadata: {
|
|
427
|
+
error: error.message,
|
|
428
|
+
operation: "evaluateVariant",
|
|
429
|
+
variant,
|
|
430
|
+
versionId
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
);
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
if (variantMetrics.length === 0) {
|
|
437
|
+
return {
|
|
438
|
+
versionId,
|
|
439
|
+
variantMetrics: [],
|
|
440
|
+
bestVariant: null,
|
|
441
|
+
insights: ["No variant performance data available"]
|
|
442
|
+
};
|
|
443
|
+
}
|
|
444
|
+
const scored = variantMetrics.map((vm) => ({
|
|
445
|
+
variant: vm.variant,
|
|
446
|
+
score: vm.metrics.successRate * 0.4 + vm.metrics.avgCompleteness / 100 * 0.4 + Math.max(0, 1 - (vm.metrics.avgDuration ?? 0) / 300) * 0.2
|
|
447
|
+
}));
|
|
448
|
+
scored.sort((a, b) => b.score - a.score);
|
|
449
|
+
const bestVariant = scored[0]?.variant ?? null;
|
|
450
|
+
const insights = [];
|
|
451
|
+
const best = bestVariant ? variantMetrics.find((vm) => vm.variant === bestVariant) : void 0;
|
|
452
|
+
if (best) {
|
|
453
|
+
insights.push(
|
|
454
|
+
`${bestVariant} variant performs best (${(best.metrics.successRate * 100).toFixed(1)}% success, ${best.metrics.avgCompleteness.toFixed(1)} avg completeness)`
|
|
455
|
+
);
|
|
456
|
+
}
|
|
457
|
+
for (const vm of variantMetrics) {
|
|
458
|
+
if (vm.variant !== bestVariant && best) {
|
|
459
|
+
const deltaSuccess = (vm.metrics.successRate - best.metrics.successRate) * 100;
|
|
460
|
+
const deltaCompleteness = vm.metrics.avgCompleteness - best.metrics.avgCompleteness;
|
|
461
|
+
if (Math.abs(deltaSuccess) > 10 || Math.abs(deltaCompleteness) > 10) {
|
|
462
|
+
insights.push(
|
|
463
|
+
`${vm.variant} shows ${deltaSuccess > 0 ? "+" : ""}${deltaSuccess.toFixed(1)}% success rate, ${deltaCompleteness > 0 ? "+" : ""}${deltaCompleteness.toFixed(1)} completeness vs ${bestVariant}`
|
|
464
|
+
);
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
return {
|
|
469
|
+
versionId,
|
|
470
|
+
variantMetrics,
|
|
471
|
+
bestVariant,
|
|
472
|
+
insights
|
|
473
|
+
};
|
|
474
|
+
} catch (error) {
|
|
475
|
+
throw new Error(
|
|
476
|
+
`Failed to compare variant performance for version ${versionId}: ${error.message}`
|
|
477
|
+
);
|
|
478
|
+
}
|
|
479
|
+
}
|
|
480
|
+
/**
|
|
481
|
+
* Close connections
|
|
482
|
+
*/
|
|
483
|
+
close() {
|
|
484
|
+
this.store.close();
|
|
485
|
+
this.telemetry.close();
|
|
486
|
+
}
|
|
487
|
+
// Private helper methods
|
|
488
|
+
/**
|
|
489
|
+
* Determine trend direction based on comparison
|
|
490
|
+
*/
|
|
491
|
+
determineTrend(recent, older, threshold = 0.05) {
|
|
492
|
+
const change = recent - older;
|
|
493
|
+
const percentChange = Math.abs(change) / older;
|
|
494
|
+
if (percentChange < threshold) {
|
|
495
|
+
return "stable";
|
|
496
|
+
} else if (change > 0) {
|
|
497
|
+
return "improving";
|
|
498
|
+
} else {
|
|
499
|
+
return "declining";
|
|
500
|
+
}
|
|
501
|
+
}
|
|
502
|
+
};
|
|
503
|
+
|
|
504
|
+
// src/prompts/optimizer.ts
|
|
505
|
+
var PromptOptimizer = class {
|
|
506
|
+
store;
|
|
507
|
+
evaluator;
|
|
508
|
+
constructor(options = {}) {
|
|
509
|
+
this.store = new PromptStore(options);
|
|
510
|
+
this.evaluator = new PromptEvaluator({
|
|
511
|
+
poeticDir: options.poeticDir,
|
|
512
|
+
telemetryLogger: options.telemetryLogger
|
|
513
|
+
});
|
|
514
|
+
}
|
|
515
|
+
/**
|
|
516
|
+
* Analyze a prompt version and generate optimization suggestions
|
|
517
|
+
*/
|
|
518
|
+
async analyze(versionId, variant) {
|
|
519
|
+
try {
|
|
520
|
+
const version = this.store.getVersion(versionId);
|
|
521
|
+
if (!version) {
|
|
522
|
+
throw new Error(`Version ${versionId} not found`);
|
|
523
|
+
}
|
|
524
|
+
const performance = await this.evaluator.calculatePerformanceMetrics(version, variant);
|
|
525
|
+
const suggestions = this.generateSuggestions(version, performance);
|
|
526
|
+
const strengths = this.identifyStrengths(performance);
|
|
527
|
+
const weaknesses = this.identifyWeaknesses(performance);
|
|
528
|
+
const overallScore = this.calculateOverallScore(performance);
|
|
529
|
+
return {
|
|
530
|
+
versionId,
|
|
531
|
+
promptId: version.promptId,
|
|
532
|
+
version: version.id,
|
|
533
|
+
performance,
|
|
534
|
+
suggestions,
|
|
535
|
+
strengths,
|
|
536
|
+
weaknesses,
|
|
537
|
+
overallScore,
|
|
538
|
+
analyzedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
539
|
+
};
|
|
540
|
+
} catch (error) {
|
|
541
|
+
throw new Error(`Failed to analyze version ${versionId}: ${error.message}`);
|
|
542
|
+
}
|
|
543
|
+
}
|
|
544
|
+
/**
|
|
545
|
+
* Compare current version with a baseline version
|
|
546
|
+
*/
|
|
547
|
+
async compareWithBaseline(currentVersionId, baselineVersionId, variant) {
|
|
548
|
+
try {
|
|
549
|
+
const currentVersion = this.store.getVersion(currentVersionId);
|
|
550
|
+
const baselineVersion = this.store.getVersion(baselineVersionId);
|
|
551
|
+
if (!currentVersion || !baselineVersion) {
|
|
552
|
+
throw new Error("One or both versions not found");
|
|
553
|
+
}
|
|
554
|
+
const currentPerf = await this.evaluator.calculatePerformanceMetrics(currentVersion, variant);
|
|
555
|
+
const baselinePerf = await this.evaluator.calculatePerformanceMetrics(
|
|
556
|
+
baselineVersion,
|
|
557
|
+
variant
|
|
558
|
+
);
|
|
559
|
+
const currentScore = this.calculateOverallScore(currentPerf);
|
|
560
|
+
const baselineScore = this.calculateOverallScore(baselinePerf);
|
|
561
|
+
const performanceChange = (currentScore - baselineScore) / baselineScore * 100;
|
|
562
|
+
const improvements = [];
|
|
563
|
+
const regressions = [];
|
|
564
|
+
const successRateChange = (currentPerf.successRate - baselinePerf.successRate) * 100;
|
|
565
|
+
if (successRateChange > 5) {
|
|
566
|
+
improvements.push(`Success rate improved by ${successRateChange.toFixed(1)}%`);
|
|
567
|
+
} else if (successRateChange < -5) {
|
|
568
|
+
regressions.push(`Success rate decreased by ${Math.abs(successRateChange).toFixed(1)}%`);
|
|
569
|
+
}
|
|
570
|
+
const completenessChange = currentPerf.avgCompleteness - baselinePerf.avgCompleteness;
|
|
571
|
+
if (completenessChange > 5) {
|
|
572
|
+
improvements.push(`Completeness improved by ${completenessChange.toFixed(1)} points`);
|
|
573
|
+
} else if (completenessChange < -5) {
|
|
574
|
+
regressions.push(
|
|
575
|
+
`Completeness decreased by ${Math.abs(completenessChange).toFixed(1)} points`
|
|
576
|
+
);
|
|
577
|
+
}
|
|
578
|
+
const durationChange = currentPerf.avgDuration - baselinePerf.avgDuration;
|
|
579
|
+
if (durationChange < -10) {
|
|
580
|
+
improvements.push(`Execution time improved by ${Math.abs(durationChange).toFixed(0)}s`);
|
|
581
|
+
} else if (durationChange > 10) {
|
|
582
|
+
regressions.push(`Execution time increased by ${durationChange.toFixed(0)}s`);
|
|
583
|
+
}
|
|
584
|
+
const deliveryChange = (currentPerf.avgCodeDeliveryRate - baselinePerf.avgCodeDeliveryRate) * 100;
|
|
585
|
+
if (deliveryChange > 5) {
|
|
586
|
+
improvements.push(`Code delivery rate improved by ${deliveryChange.toFixed(1)}%`);
|
|
587
|
+
} else if (deliveryChange < -5) {
|
|
588
|
+
regressions.push(`Code delivery rate decreased by ${Math.abs(deliveryChange).toFixed(1)}%`);
|
|
589
|
+
}
|
|
590
|
+
let recommendation;
|
|
591
|
+
if (performanceChange > 10 && regressions.length === 0) {
|
|
592
|
+
recommendation = "Significant improvement - recommend activating this version";
|
|
593
|
+
} else if (performanceChange > 5) {
|
|
594
|
+
recommendation = "Moderate improvement - consider activating after review";
|
|
595
|
+
} else if (performanceChange < -10) {
|
|
596
|
+
recommendation = "Performance regression - do not activate, investigate issues";
|
|
597
|
+
} else if (performanceChange < -5) {
|
|
598
|
+
recommendation = "Minor regression - investigate before activating";
|
|
599
|
+
} else {
|
|
600
|
+
recommendation = "Performance is similar - review improvements and regressions before deciding";
|
|
601
|
+
}
|
|
602
|
+
return {
|
|
603
|
+
currentVersion: currentVersionId,
|
|
604
|
+
baselineVersion: baselineVersionId,
|
|
605
|
+
performanceChange,
|
|
606
|
+
improvements,
|
|
607
|
+
regressions,
|
|
608
|
+
recommendation
|
|
609
|
+
};
|
|
610
|
+
} catch (error) {
|
|
611
|
+
throw new Error(`Failed to compare versions: ${error.message}`);
|
|
612
|
+
}
|
|
613
|
+
}
|
|
614
|
+
/**
|
|
615
|
+
* Generate optimization suggestions for all versions of a prompt
|
|
616
|
+
*/
|
|
617
|
+
async analyzeAllVersions(promptId, variant) {
|
|
618
|
+
try {
|
|
619
|
+
const versions = this.store.listVersions(promptId);
|
|
620
|
+
const analyses = [];
|
|
621
|
+
for (const version of versions) {
|
|
622
|
+
try {
|
|
623
|
+
const analysis = await this.analyze(version.id, variant);
|
|
624
|
+
analyses.push(analysis);
|
|
625
|
+
} catch (error) {
|
|
626
|
+
emitWarning(
|
|
627
|
+
"PR-001",
|
|
628
|
+
{
|
|
629
|
+
taskId: "prompt-optimizer",
|
|
630
|
+
subsystem: "prompts",
|
|
631
|
+
executionPhase: "execution"
|
|
632
|
+
},
|
|
633
|
+
{
|
|
634
|
+
severity: "WARN" /* WARN */,
|
|
635
|
+
metadata: {
|
|
636
|
+
error: error.message,
|
|
637
|
+
operation: "analyzeVersion",
|
|
638
|
+
versionId: version.id,
|
|
639
|
+
promptId
|
|
640
|
+
}
|
|
641
|
+
}
|
|
642
|
+
);
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
return analyses;
|
|
646
|
+
} catch (error) {
|
|
647
|
+
throw new Error(
|
|
648
|
+
`Failed to analyze versions for prompt ${promptId}: ${error.message}`
|
|
649
|
+
);
|
|
650
|
+
}
|
|
651
|
+
}
|
|
652
|
+
/**
|
|
653
|
+
* Recommend the best version based on performance
|
|
654
|
+
*/
|
|
655
|
+
async recommendBestVersion(promptId, variant) {
|
|
656
|
+
try {
|
|
657
|
+
const analyses = await this.analyzeAllVersions(promptId, variant);
|
|
658
|
+
if (analyses.length === 0) {
|
|
659
|
+
throw new Error("No versions found for prompt");
|
|
660
|
+
}
|
|
661
|
+
const validAnalyses = analyses.filter((a) => a.performance.sampleSize >= 5);
|
|
662
|
+
if (validAnalyses.length === 0) {
|
|
663
|
+
throw new Error("No versions have sufficient performance data");
|
|
664
|
+
}
|
|
665
|
+
validAnalyses.sort((a, b) => b.overallScore - a.overallScore);
|
|
666
|
+
const best = validAnalyses[0];
|
|
667
|
+
if (!best) {
|
|
668
|
+
throw new Error("No best version found");
|
|
669
|
+
}
|
|
670
|
+
const rationale = this.generateRecommendationRationale(best);
|
|
671
|
+
const alternatives = validAnalyses.slice(1, 4).map((analysis) => ({
|
|
672
|
+
version: analysis.versionId,
|
|
673
|
+
reason: this.generateAlternativeReason(analysis, best)
|
|
674
|
+
}));
|
|
675
|
+
return {
|
|
676
|
+
recommendedVersion: best.versionId,
|
|
677
|
+
rationale,
|
|
678
|
+
alternatives
|
|
679
|
+
};
|
|
680
|
+
} catch (error) {
|
|
681
|
+
throw new Error(`Failed to recommend best version: ${error.message}`);
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
/**
|
|
685
|
+
* Close connections
|
|
686
|
+
*/
|
|
687
|
+
close() {
|
|
688
|
+
this.store.close();
|
|
689
|
+
this.evaluator.close();
|
|
690
|
+
}
|
|
691
|
+
// Private helper methods
|
|
692
|
+
/**
|
|
693
|
+
* Generate optimization suggestions based on performance
|
|
694
|
+
*/
|
|
695
|
+
generateSuggestions(version, performance) {
|
|
696
|
+
const suggestions = [];
|
|
697
|
+
if (performance.successRate < 0.7) {
|
|
698
|
+
suggestions.push({
|
|
699
|
+
promptId: version.promptId,
|
|
700
|
+
versionId: version.id,
|
|
701
|
+
type: "improve_clarity",
|
|
702
|
+
priority: "high",
|
|
703
|
+
description: "Low success rate indicates unclear or ambiguous instructions",
|
|
704
|
+
rationale: `Current success rate is ${(performance.successRate * 100).toFixed(1)}% (target: 70%+)`,
|
|
705
|
+
affectedMetrics: ["successRate"],
|
|
706
|
+
expectedImprovement: "Clearer instructions should increase success rate by 15-20%",
|
|
707
|
+
suggestedChanges: "Add explicit step-by-step instructions, clarify expected outcomes, provide examples",
|
|
708
|
+
estimatedEffort: "medium",
|
|
709
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
710
|
+
generatedBy: "analyzer"
|
|
711
|
+
});
|
|
712
|
+
}
|
|
713
|
+
if (performance.avgCompleteness < 70) {
|
|
714
|
+
suggestions.push({
|
|
715
|
+
promptId: version.promptId,
|
|
716
|
+
versionId: version.id,
|
|
717
|
+
type: "add_context",
|
|
718
|
+
priority: "high",
|
|
719
|
+
description: "Low completeness suggests insufficient context or guidance",
|
|
720
|
+
rationale: `Average completeness is ${performance.avgCompleteness.toFixed(1)} (target: 70+)`,
|
|
721
|
+
affectedMetrics: ["avgCompleteness"],
|
|
722
|
+
expectedImprovement: "Better context should increase completeness by 10-15 points",
|
|
723
|
+
suggestedChanges: "Add more context about requirements, provide acceptance criteria, include edge cases",
|
|
724
|
+
estimatedEffort: "medium",
|
|
725
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
726
|
+
generatedBy: "analyzer"
|
|
727
|
+
});
|
|
728
|
+
}
|
|
729
|
+
if (performance.stdDevCompleteness > 20) {
|
|
730
|
+
suggestions.push({
|
|
731
|
+
promptId: version.promptId,
|
|
732
|
+
versionId: version.id,
|
|
733
|
+
type: "improve_clarity",
|
|
734
|
+
priority: "medium",
|
|
735
|
+
description: "High variability in completeness indicates inconsistent interpretation",
|
|
736
|
+
rationale: `Standard deviation of ${performance.stdDevCompleteness.toFixed(1)} suggests prompt ambiguity`,
|
|
737
|
+
affectedMetrics: ["stdDevCompleteness"],
|
|
738
|
+
expectedImprovement: "More specific instructions should reduce variability by 30-40%",
|
|
739
|
+
suggestedChanges: "Make instructions more specific, reduce ambiguous terms, add constraints",
|
|
740
|
+
estimatedEffort: "low",
|
|
741
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
742
|
+
generatedBy: "analyzer"
|
|
743
|
+
});
|
|
744
|
+
}
|
|
745
|
+
if (performance.avgDuration > 180) {
|
|
746
|
+
suggestions.push({
|
|
747
|
+
promptId: version.promptId,
|
|
748
|
+
versionId: version.id,
|
|
749
|
+
type: "split_prompt",
|
|
750
|
+
priority: "medium",
|
|
751
|
+
description: "Long execution time suggests prompt may be too complex",
|
|
752
|
+
rationale: `Average duration of ${performance.avgDuration.toFixed(0)}s exceeds 3 minutes`,
|
|
753
|
+
affectedMetrics: ["avgDuration"],
|
|
754
|
+
expectedImprovement: "Breaking into smaller tasks should reduce duration by 40-50%",
|
|
755
|
+
suggestedChanges: "Split into multiple focused prompts, reduce scope, prioritize critical features",
|
|
756
|
+
estimatedEffort: "high",
|
|
757
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
758
|
+
generatedBy: "analyzer"
|
|
759
|
+
});
|
|
760
|
+
}
|
|
761
|
+
if (performance.avgCodeDeliveryRate < 0.8) {
|
|
762
|
+
suggestions.push({
|
|
763
|
+
promptId: version.promptId,
|
|
764
|
+
versionId: version.id,
|
|
765
|
+
type: "update_strategy",
|
|
766
|
+
priority: "medium",
|
|
767
|
+
description: "Low code delivery rate suggests implementation difficulties",
|
|
768
|
+
rationale: `Code delivery rate of ${(performance.avgCodeDeliveryRate * 100).toFixed(1)}% is below 80%`,
|
|
769
|
+
affectedMetrics: ["avgCodeDeliveryRate"],
|
|
770
|
+
expectedImprovement: "Better guidance should increase delivery rate by 10-15%",
|
|
771
|
+
suggestedChanges: "Add implementation hints, provide code patterns, clarify technical requirements",
|
|
772
|
+
estimatedEffort: "medium",
|
|
773
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
774
|
+
generatedBy: "analyzer"
|
|
775
|
+
});
|
|
776
|
+
}
|
|
777
|
+
const wordCount = version.content.split(/\s+/).length;
|
|
778
|
+
if (wordCount < 50 && performance.avgCompleteness < 85) {
|
|
779
|
+
suggestions.push({
|
|
780
|
+
promptId: version.promptId,
|
|
781
|
+
versionId: version.id,
|
|
782
|
+
type: "add_context",
|
|
783
|
+
priority: "low",
|
|
784
|
+
description: "Short prompt may lack sufficient detail",
|
|
785
|
+
rationale: `Prompt has only ${wordCount} words - may need more context`,
|
|
786
|
+
affectedMetrics: ["avgCompleteness"],
|
|
787
|
+
expectedImprovement: "More detailed prompts typically improve completeness by 10-20%",
|
|
788
|
+
suggestedChanges: "Add background information, specify requirements, provide examples",
|
|
789
|
+
estimatedEffort: "low",
|
|
790
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
791
|
+
generatedBy: "analyzer"
|
|
792
|
+
});
|
|
793
|
+
} else if (wordCount > 500) {
|
|
794
|
+
suggestions.push({
|
|
795
|
+
promptId: version.promptId,
|
|
796
|
+
versionId: version.id,
|
|
797
|
+
type: "remove_redundancy",
|
|
798
|
+
priority: "low",
|
|
799
|
+
description: "Long prompt may contain redundancy or unnecessary detail",
|
|
800
|
+
rationale: `Prompt has ${wordCount} words - may overwhelm or confuse`,
|
|
801
|
+
affectedMetrics: ["avgDuration", "successRate"],
|
|
802
|
+
expectedImprovement: "Concise prompts can improve clarity and reduce execution time",
|
|
803
|
+
suggestedChanges: "Remove redundant information, consolidate instructions, focus on essentials",
|
|
804
|
+
estimatedEffort: "medium",
|
|
805
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
806
|
+
generatedBy: "analyzer"
|
|
807
|
+
});
|
|
808
|
+
}
|
|
809
|
+
return suggestions;
|
|
810
|
+
}
|
|
811
|
+
/**
|
|
812
|
+
* Identify strengths of a prompt based on performance
|
|
813
|
+
*/
|
|
814
|
+
identifyStrengths(performance) {
|
|
815
|
+
const strengths = [];
|
|
816
|
+
if (performance.successRate >= 0.85) {
|
|
817
|
+
strengths.push(`Excellent success rate (${(performance.successRate * 100).toFixed(1)}%)`);
|
|
818
|
+
} else if (performance.successRate >= 0.7) {
|
|
819
|
+
strengths.push(`Good success rate (${(performance.successRate * 100).toFixed(1)}%)`);
|
|
820
|
+
}
|
|
821
|
+
if (performance.avgCompleteness >= 85) {
|
|
822
|
+
strengths.push(`High completeness (${performance.avgCompleteness.toFixed(1)})`);
|
|
823
|
+
} else if (performance.avgCompleteness >= 70) {
|
|
824
|
+
strengths.push(`Good completeness (${performance.avgCompleteness.toFixed(1)})`);
|
|
825
|
+
}
|
|
826
|
+
if (performance.avgCodeDeliveryRate >= 0.9) {
|
|
827
|
+
strengths.push(
|
|
828
|
+
`Excellent code delivery rate (${(performance.avgCodeDeliveryRate * 100).toFixed(1)}%)`
|
|
829
|
+
);
|
|
830
|
+
}
|
|
831
|
+
if (performance.avgDuration <= 60) {
|
|
832
|
+
strengths.push(`Fast execution (${performance.avgDuration.toFixed(0)}s avg)`);
|
|
833
|
+
} else if (performance.avgDuration <= 120) {
|
|
834
|
+
strengths.push(`Good execution time (${performance.avgDuration.toFixed(0)}s avg)`);
|
|
835
|
+
}
|
|
836
|
+
if (performance.stdDevCompleteness < 10) {
|
|
837
|
+
strengths.push("Consistent performance across executions");
|
|
838
|
+
}
|
|
839
|
+
if (strengths.length === 0) {
|
|
840
|
+
strengths.push("Meets basic quality standards");
|
|
841
|
+
}
|
|
842
|
+
return strengths;
|
|
843
|
+
}
|
|
844
|
+
/**
|
|
845
|
+
* Identify weaknesses of a prompt based on performance
|
|
846
|
+
*/
|
|
847
|
+
identifyWeaknesses(performance) {
|
|
848
|
+
const weaknesses = [];
|
|
849
|
+
if (performance.successRate < 0.5) {
|
|
850
|
+
weaknesses.push(`Poor success rate (${(performance.successRate * 100).toFixed(1)}%)`);
|
|
851
|
+
} else if (performance.successRate < 0.7) {
|
|
852
|
+
weaknesses.push(`Below-target success rate (${(performance.successRate * 100).toFixed(1)}%)`);
|
|
853
|
+
}
|
|
854
|
+
if (performance.avgCompleteness < 60) {
|
|
855
|
+
weaknesses.push(`Low completeness (${performance.avgCompleteness.toFixed(1)})`);
|
|
856
|
+
} else if (performance.avgCompleteness < 70) {
|
|
857
|
+
weaknesses.push(`Below-target completeness (${performance.avgCompleteness.toFixed(1)})`);
|
|
858
|
+
}
|
|
859
|
+
if (performance.avgCodeDeliveryRate < 0.7) {
|
|
860
|
+
weaknesses.push(
|
|
861
|
+
`Low code delivery rate (${(performance.avgCodeDeliveryRate * 100).toFixed(1)}%)`
|
|
862
|
+
);
|
|
863
|
+
}
|
|
864
|
+
if (performance.avgDuration > 300) {
|
|
865
|
+
weaknesses.push(`Slow execution (${performance.avgDuration.toFixed(0)}s avg)`);
|
|
866
|
+
} else if (performance.avgDuration > 180) {
|
|
867
|
+
weaknesses.push(`Long execution time (${performance.avgDuration.toFixed(0)}s avg)`);
|
|
868
|
+
}
|
|
869
|
+
if (performance.stdDevCompleteness > 25) {
|
|
870
|
+
weaknesses.push("Inconsistent performance across executions");
|
|
871
|
+
}
|
|
872
|
+
if (weaknesses.length === 0) {
|
|
873
|
+
weaknesses.push("No significant weaknesses identified");
|
|
874
|
+
}
|
|
875
|
+
return weaknesses;
|
|
876
|
+
}
|
|
877
|
+
/**
|
|
878
|
+
* Calculate overall performance score (0-100)
|
|
879
|
+
*/
|
|
880
|
+
calculateOverallScore(performance) {
|
|
881
|
+
const successScore = performance.successRate * 100 * 0.35;
|
|
882
|
+
const completenessScore = performance.avgCompleteness * 0.35;
|
|
883
|
+
const durationScore = Math.max(0, 1 - performance.avgDuration / 300) * 100 * 0.15;
|
|
884
|
+
const deliveryScore = performance.avgCodeDeliveryRate * 100 * 0.15;
|
|
885
|
+
return successScore + completenessScore + durationScore + deliveryScore;
|
|
886
|
+
}
|
|
887
|
+
/**
|
|
888
|
+
* Generate rationale for recommendation
|
|
889
|
+
*/
|
|
890
|
+
generateRecommendationRationale(analysis) {
|
|
891
|
+
const parts = [];
|
|
892
|
+
parts.push(`Overall performance score: ${analysis.overallScore.toFixed(1)}/100`);
|
|
893
|
+
parts.push(`Success rate: ${(analysis.performance.successRate * 100).toFixed(1)}%`);
|
|
894
|
+
parts.push(`Average completeness: ${analysis.performance.avgCompleteness.toFixed(1)}`);
|
|
895
|
+
parts.push(`Sample size: ${analysis.performance.sampleSize} executions`);
|
|
896
|
+
if (analysis.strengths.length > 0) {
|
|
897
|
+
parts.push(`Key strengths: ${analysis.strengths.slice(0, 2).join(", ")}`);
|
|
898
|
+
}
|
|
899
|
+
return parts.join(". ");
|
|
900
|
+
}
|
|
901
|
+
/**
|
|
902
|
+
* Generate reason why version is an alternative
|
|
903
|
+
*/
|
|
904
|
+
generateAlternativeReason(analysis, best) {
|
|
905
|
+
const scoreDiff = best.overallScore - analysis.overallScore;
|
|
906
|
+
if (scoreDiff < 5) {
|
|
907
|
+
return `Similar performance (score: ${analysis.overallScore.toFixed(1)}, only ${scoreDiff.toFixed(1)} points lower)`;
|
|
908
|
+
} else if (analysis.performance.avgDuration < best.performance.avgDuration) {
|
|
909
|
+
return `Faster execution (${analysis.performance.avgDuration.toFixed(0)}s vs ${best.performance.avgDuration.toFixed(0)}s)`;
|
|
910
|
+
} else if (analysis.performance.stdDevCompleteness < best.performance.stdDevCompleteness) {
|
|
911
|
+
return "More consistent results";
|
|
912
|
+
} else {
|
|
913
|
+
return `Lower score (${analysis.overallScore.toFixed(1)}) but may suit different use cases`;
|
|
914
|
+
}
|
|
915
|
+
}
|
|
916
|
+
};
|
|
917
|
+
|
|
918
|
+
// src/prompts/auto-tuner.ts
|
|
919
|
+
var AutoTuningEngine = class {
|
|
920
|
+
manager;
|
|
921
|
+
evaluator;
|
|
922
|
+
optimizer;
|
|
923
|
+
options;
|
|
924
|
+
// Track activations and rollbacks
|
|
925
|
+
activationHistory = /* @__PURE__ */ new Map();
|
|
926
|
+
rollbackCounts = /* @__PURE__ */ new Map();
|
|
927
|
+
constructor(options = {}) {
|
|
928
|
+
this.manager = new PromptManager(options);
|
|
929
|
+
this.evaluator = new PromptEvaluator(options);
|
|
930
|
+
this.optimizer = new PromptOptimizer(options);
|
|
931
|
+
this.options = {
|
|
932
|
+
poeticDir: resolvePoeticDirFromContext(options.poeticDir),
|
|
933
|
+
minSampleSize: options.minSampleSize ?? 10,
|
|
934
|
+
significanceThreshold: options.significanceThreshold ?? 10,
|
|
935
|
+
confidenceLevel: options.confidenceLevel ?? 0.8,
|
|
936
|
+
enableAutoActivation: options.enableAutoActivation ?? false,
|
|
937
|
+
maxRollbacks: options.maxRollbacks ?? 3
|
|
938
|
+
};
|
|
939
|
+
}
|
|
940
|
+
/**
|
|
941
|
+
* Analyze whether a prompt should be auto-tuned
|
|
942
|
+
*/
|
|
943
|
+
async analyzeForActivation(promptId) {
|
|
944
|
+
try {
|
|
945
|
+
this.manager.getOrThrow(promptId);
|
|
946
|
+
const currentVersion = this.manager.getActiveVersionOrThrow(promptId);
|
|
947
|
+
const rollbackCount = this.rollbackCounts.get(promptId) || 0;
|
|
948
|
+
if (rollbackCount >= this.options.maxRollbacks) {
|
|
949
|
+
return {
|
|
950
|
+
shouldActivate: false,
|
|
951
|
+
candidateVersionId: null,
|
|
952
|
+
currentVersionId: currentVersion.id,
|
|
953
|
+
improvementPercentage: 0,
|
|
954
|
+
statisticalSignificance: 0,
|
|
955
|
+
confidence: 0,
|
|
956
|
+
safetyChecks: [
|
|
957
|
+
{
|
|
958
|
+
check: "rollback_limit",
|
|
959
|
+
passed: false,
|
|
960
|
+
message: `Maximum rollbacks (${this.options.maxRollbacks}) reached for this prompt`
|
|
961
|
+
}
|
|
962
|
+
],
|
|
963
|
+
recommendation: "reject",
|
|
964
|
+
rationale: "Auto-tuning disabled due to excessive rollbacks",
|
|
965
|
+
decidedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
966
|
+
};
|
|
967
|
+
}
|
|
968
|
+
const currentPerf = await this.evaluator.calculatePerformanceMetrics(currentVersion);
|
|
969
|
+
let bestVersionResult;
|
|
970
|
+
let candidateVersionId;
|
|
971
|
+
try {
|
|
972
|
+
bestVersionResult = await this.optimizer.recommendBestVersion(promptId);
|
|
973
|
+
candidateVersionId = bestVersionResult.recommendedVersion;
|
|
974
|
+
} catch (optimizerError) {
|
|
975
|
+
if (optimizerError.message.includes("No versions have sufficient performance data")) {
|
|
976
|
+
return {
|
|
977
|
+
shouldActivate: false,
|
|
978
|
+
candidateVersionId: null,
|
|
979
|
+
currentVersionId: currentVersion.id,
|
|
980
|
+
improvementPercentage: 0,
|
|
981
|
+
statisticalSignificance: 0,
|
|
982
|
+
confidence: 0,
|
|
983
|
+
safetyChecks: [
|
|
984
|
+
{
|
|
985
|
+
check: "sample_size",
|
|
986
|
+
passed: false,
|
|
987
|
+
message: "No versions have sufficient performance data (need at least 5 samples)"
|
|
988
|
+
}
|
|
989
|
+
],
|
|
990
|
+
recommendation: "wait_for_data",
|
|
991
|
+
rationale: "Need more execution data before any version comparison",
|
|
992
|
+
decidedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
993
|
+
};
|
|
994
|
+
}
|
|
995
|
+
throw optimizerError;
|
|
996
|
+
}
|
|
997
|
+
if (currentPerf.sampleSize < this.options.minSampleSize) {
|
|
998
|
+
return {
|
|
999
|
+
shouldActivate: false,
|
|
1000
|
+
candidateVersionId: null,
|
|
1001
|
+
currentVersionId: currentVersion.id,
|
|
1002
|
+
improvementPercentage: 0,
|
|
1003
|
+
statisticalSignificance: 0,
|
|
1004
|
+
confidence: 0,
|
|
1005
|
+
safetyChecks: [
|
|
1006
|
+
{
|
|
1007
|
+
check: "sample_size",
|
|
1008
|
+
passed: false,
|
|
1009
|
+
message: `Insufficient data (${currentPerf.sampleSize}/${this.options.minSampleSize} samples)`
|
|
1010
|
+
}
|
|
1011
|
+
],
|
|
1012
|
+
recommendation: "wait_for_data",
|
|
1013
|
+
rationale: "Need more execution data before considering activation",
|
|
1014
|
+
decidedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
1015
|
+
};
|
|
1016
|
+
}
|
|
1017
|
+
if (candidateVersionId === currentVersion.id) {
|
|
1018
|
+
return {
|
|
1019
|
+
shouldActivate: false,
|
|
1020
|
+
candidateVersionId: null,
|
|
1021
|
+
currentVersionId: currentVersion.id,
|
|
1022
|
+
improvementPercentage: 0,
|
|
1023
|
+
statisticalSignificance: 1,
|
|
1024
|
+
confidence: 1,
|
|
1025
|
+
safetyChecks: [
|
|
1026
|
+
{
|
|
1027
|
+
check: "current_is_best",
|
|
1028
|
+
passed: true,
|
|
1029
|
+
message: "Current version is already the best performer"
|
|
1030
|
+
}
|
|
1031
|
+
],
|
|
1032
|
+
recommendation: "monitor",
|
|
1033
|
+
rationale: "Current version performs optimally",
|
|
1034
|
+
decidedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
1035
|
+
};
|
|
1036
|
+
}
|
|
1037
|
+
const candidateVersion = this.manager.getVersion(promptId, candidateVersionId);
|
|
1038
|
+
if (!candidateVersion) {
|
|
1039
|
+
throw new Error(`Candidate version ${candidateVersionId} not found`);
|
|
1040
|
+
}
|
|
1041
|
+
const candidatePerf = await this.evaluator.calculatePerformanceMetrics(candidateVersion);
|
|
1042
|
+
const currentScore = this.calculateScore(currentPerf);
|
|
1043
|
+
const candidateScore = this.calculateScore(candidatePerf);
|
|
1044
|
+
const improvementPercentage = (candidateScore - currentScore) / currentScore * 100;
|
|
1045
|
+
const safetyChecks = await this.runSafetyChecks(
|
|
1046
|
+
currentPerf,
|
|
1047
|
+
candidatePerf,
|
|
1048
|
+
improvementPercentage
|
|
1049
|
+
);
|
|
1050
|
+
const allChecksPassed = safetyChecks.every((c) => c.passed);
|
|
1051
|
+
const statisticalSignificance = this.calculateStatisticalSignificance(
|
|
1052
|
+
currentPerf,
|
|
1053
|
+
candidatePerf
|
|
1054
|
+
);
|
|
1055
|
+
const confidence = this.calculateConfidence(
|
|
1056
|
+
candidatePerf.sampleSize,
|
|
1057
|
+
statisticalSignificance,
|
|
1058
|
+
improvementPercentage
|
|
1059
|
+
);
|
|
1060
|
+
let shouldActivate = false;
|
|
1061
|
+
let recommendation;
|
|
1062
|
+
let rationale;
|
|
1063
|
+
if (!allChecksPassed) {
|
|
1064
|
+
recommendation = "reject";
|
|
1065
|
+
rationale = `Safety checks failed: ${safetyChecks.filter((c) => !c.passed).map((c) => c.check).join(", ")}`;
|
|
1066
|
+
} else if (candidatePerf.sampleSize < this.options.minSampleSize) {
|
|
1067
|
+
recommendation = "wait_for_data";
|
|
1068
|
+
rationale = "Candidate needs more testing before activation";
|
|
1069
|
+
} else if (improvementPercentage < this.options.significanceThreshold) {
|
|
1070
|
+
recommendation = "monitor";
|
|
1071
|
+
rationale = `Improvement (${improvementPercentage.toFixed(1)}%) below threshold (${this.options.significanceThreshold}%)`;
|
|
1072
|
+
} else if (confidence < this.options.confidenceLevel) {
|
|
1073
|
+
recommendation = "monitor";
|
|
1074
|
+
rationale = `Confidence (${confidence.toFixed(2)}) below threshold (${this.options.confidenceLevel})`;
|
|
1075
|
+
} else if (this.options.enableAutoActivation) {
|
|
1076
|
+
shouldActivate = true;
|
|
1077
|
+
recommendation = "activate";
|
|
1078
|
+
rationale = `Significant improvement (${improvementPercentage.toFixed(1)}%) with high confidence (${confidence.toFixed(2)})`;
|
|
1079
|
+
} else {
|
|
1080
|
+
recommendation = "monitor";
|
|
1081
|
+
rationale = "Auto-activation disabled - manual review required";
|
|
1082
|
+
}
|
|
1083
|
+
return {
|
|
1084
|
+
shouldActivate,
|
|
1085
|
+
candidateVersionId,
|
|
1086
|
+
currentVersionId: currentVersion.id,
|
|
1087
|
+
improvementPercentage,
|
|
1088
|
+
statisticalSignificance,
|
|
1089
|
+
confidence,
|
|
1090
|
+
safetyChecks,
|
|
1091
|
+
recommendation,
|
|
1092
|
+
rationale,
|
|
1093
|
+
decidedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
1094
|
+
};
|
|
1095
|
+
} catch (error) {
|
|
1096
|
+
throw new Error(`Failed to analyze prompt for activation: ${error.message}`);
|
|
1097
|
+
}
|
|
1098
|
+
}
|
|
1099
|
+
/**
|
|
1100
|
+
* Execute activation if decision recommends it
|
|
1101
|
+
*/
|
|
1102
|
+
async executeActivation(promptId, decision, options = {}) {
|
|
1103
|
+
try {
|
|
1104
|
+
if (!decision.shouldActivate || !decision.candidateVersionId) {
|
|
1105
|
+
return null;
|
|
1106
|
+
}
|
|
1107
|
+
const currentVersion = this.manager.getVersion(promptId, decision.currentVersionId);
|
|
1108
|
+
const candidateVersion = this.manager.getVersion(promptId, decision.candidateVersionId);
|
|
1109
|
+
if (!currentVersion || !candidateVersion) {
|
|
1110
|
+
throw new Error("Version not found");
|
|
1111
|
+
}
|
|
1112
|
+
const previousPerf = await this.evaluator.calculatePerformanceMetrics(currentVersion);
|
|
1113
|
+
const expectedPerf = await this.evaluator.calculatePerformanceMetrics(candidateVersion);
|
|
1114
|
+
await this.manager.activateVersion(decision.candidateVersionId, {
|
|
1115
|
+
reason: decision.rationale,
|
|
1116
|
+
skipSafetyChecks: options.skipSafetyChecks
|
|
1117
|
+
});
|
|
1118
|
+
const record = {
|
|
1119
|
+
id: this.generateActivationId(),
|
|
1120
|
+
promptId,
|
|
1121
|
+
previousVersionId: decision.currentVersionId,
|
|
1122
|
+
newVersionId: decision.candidateVersionId,
|
|
1123
|
+
previousPerformance: previousPerf,
|
|
1124
|
+
expectedPerformance: expectedPerf,
|
|
1125
|
+
activatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1126
|
+
reason: decision.rationale,
|
|
1127
|
+
confidence: decision.confidence,
|
|
1128
|
+
monitoringStatus: "active"
|
|
1129
|
+
};
|
|
1130
|
+
this.activationHistory.set(record.id, record);
|
|
1131
|
+
return record;
|
|
1132
|
+
} catch (error) {
|
|
1133
|
+
throw new Error(`Failed to execute activation: ${error.message}`);
|
|
1134
|
+
}
|
|
1135
|
+
}
|
|
1136
|
+
/**
|
|
1137
|
+
* Monitor activated prompts for regressions
|
|
1138
|
+
*/
|
|
1139
|
+
async monitorActivation(activationId, minSamplesSinceActivation = 5) {
|
|
1140
|
+
try {
|
|
1141
|
+
const record = this.activationHistory.get(activationId);
|
|
1142
|
+
if (!record) {
|
|
1143
|
+
throw new Error(`Activation record ${activationId} not found`);
|
|
1144
|
+
}
|
|
1145
|
+
if (record.monitoringStatus !== "active") {
|
|
1146
|
+
return {
|
|
1147
|
+
status: record.monitoringStatus === "validated" ? "validated" : "needs_rollback",
|
|
1148
|
+
reason: `Already ${record.monitoringStatus}`,
|
|
1149
|
+
shouldRollback: false
|
|
1150
|
+
};
|
|
1151
|
+
}
|
|
1152
|
+
const currentVersion = this.manager.getVersion(record.promptId, record.newVersionId);
|
|
1153
|
+
if (!currentVersion) {
|
|
1154
|
+
throw new Error(`Version ${record.newVersionId} not found`);
|
|
1155
|
+
}
|
|
1156
|
+
const currentPerf = await this.evaluator.calculatePerformanceMetrics(currentVersion);
|
|
1157
|
+
const samplesSinceActivation = currentPerf.sampleSize - record.expectedPerformance.sampleSize;
|
|
1158
|
+
if (samplesSinceActivation < minSamplesSinceActivation) {
|
|
1159
|
+
return {
|
|
1160
|
+
status: "validating",
|
|
1161
|
+
reason: `Still collecting data (${samplesSinceActivation}/${minSamplesSinceActivation} samples)`,
|
|
1162
|
+
shouldRollback: false
|
|
1163
|
+
};
|
|
1164
|
+
}
|
|
1165
|
+
const expectedScore = this.calculateScore(record.expectedPerformance);
|
|
1166
|
+
const actualScore = this.calculateScore(currentPerf);
|
|
1167
|
+
const performanceChange = (actualScore - expectedScore) / expectedScore * 100;
|
|
1168
|
+
if (performanceChange < -15) {
|
|
1169
|
+
return {
|
|
1170
|
+
status: "needs_rollback",
|
|
1171
|
+
reason: `Significant regression detected (${performanceChange.toFixed(1)}% worse than expected)`,
|
|
1172
|
+
shouldRollback: true
|
|
1173
|
+
};
|
|
1174
|
+
} else if (performanceChange >= -5) {
|
|
1175
|
+
record.monitoringStatus = "validated";
|
|
1176
|
+
record.validatedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
1177
|
+
this.activationHistory.set(activationId, record);
|
|
1178
|
+
return {
|
|
1179
|
+
status: "validated",
|
|
1180
|
+
reason: `Performance validated (${performanceChange >= 0 ? "+" : ""}${performanceChange.toFixed(1)}%)`,
|
|
1181
|
+
shouldRollback: false
|
|
1182
|
+
};
|
|
1183
|
+
} else {
|
|
1184
|
+
return {
|
|
1185
|
+
status: "validating",
|
|
1186
|
+
reason: `Minor deviation (${performanceChange.toFixed(1)}%), continuing to monitor`,
|
|
1187
|
+
shouldRollback: false
|
|
1188
|
+
};
|
|
1189
|
+
}
|
|
1190
|
+
} catch (error) {
|
|
1191
|
+
throw new Error(`Failed to monitor activation: ${error.message}`);
|
|
1192
|
+
}
|
|
1193
|
+
}
|
|
1194
|
+
/**
|
|
1195
|
+
* Rollback an activation if it performed poorly
|
|
1196
|
+
*/
|
|
1197
|
+
async rollbackActivation(activationId, reason) {
|
|
1198
|
+
try {
|
|
1199
|
+
const record = this.activationHistory.get(activationId);
|
|
1200
|
+
if (!record) {
|
|
1201
|
+
throw new Error(`Activation record ${activationId} not found`);
|
|
1202
|
+
}
|
|
1203
|
+
if (record.monitoringStatus === "rolled_back") {
|
|
1204
|
+
throw new Error("Activation already rolled back");
|
|
1205
|
+
}
|
|
1206
|
+
await this.manager.activateVersion(record.previousVersionId);
|
|
1207
|
+
record.monitoringStatus = "rolled_back";
|
|
1208
|
+
record.rolledBackAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
1209
|
+
record.rollbackReason = reason;
|
|
1210
|
+
this.activationHistory.set(activationId, record);
|
|
1211
|
+
const currentCount = this.rollbackCounts.get(record.promptId) || 0;
|
|
1212
|
+
this.rollbackCounts.set(record.promptId, currentCount + 1);
|
|
1213
|
+
} catch (error) {
|
|
1214
|
+
throw new Error(`Failed to rollback activation: ${error.message}`);
|
|
1215
|
+
}
|
|
1216
|
+
}
|
|
1217
|
+
/**
|
|
1218
|
+
* Get activation history for a prompt.
|
|
1219
|
+
*
|
|
1220
|
+
* Returns the in-memory ActivationRecord set (richer: includes previous
|
|
1221
|
+
* performance, monitoring status). Falls back to the persisted
|
|
1222
|
+
* activation_history table for activations that happened in earlier
|
|
1223
|
+
* processes, surfacing them as minimal records.
|
|
1224
|
+
*/
|
|
1225
|
+
getActivationHistory(promptId) {
|
|
1226
|
+
const inMemory = Array.from(this.activationHistory.values()).filter(
|
|
1227
|
+
(record) => record.promptId === promptId
|
|
1228
|
+
);
|
|
1229
|
+
const seen = new Set(inMemory.map((r) => r.id));
|
|
1230
|
+
let persisted = [];
|
|
1231
|
+
try {
|
|
1232
|
+
const store = this.manager.getStore();
|
|
1233
|
+
if (store && typeof store.listActivationHistory === "function") {
|
|
1234
|
+
const rows = store.listActivationHistory(promptId);
|
|
1235
|
+
const emptyMetrics = (versionId) => ({
|
|
1236
|
+
promptId,
|
|
1237
|
+
versionId,
|
|
1238
|
+
totalExecutions: 0,
|
|
1239
|
+
successfulExecutions: 0,
|
|
1240
|
+
failedExecutions: 0,
|
|
1241
|
+
successRate: 0,
|
|
1242
|
+
avgCompleteness: 0,
|
|
1243
|
+
avgCodeDeliveryRate: 0,
|
|
1244
|
+
avgDuration: 0,
|
|
1245
|
+
minCompleteness: 0,
|
|
1246
|
+
maxCompleteness: 0,
|
|
1247
|
+
stdDevCompleteness: 0,
|
|
1248
|
+
sampleSize: 0,
|
|
1249
|
+
firstExecutionAt: "",
|
|
1250
|
+
lastExecutionAt: "",
|
|
1251
|
+
calculatedAt: ""
|
|
1252
|
+
});
|
|
1253
|
+
persisted = rows.filter((row) => !seen.has(row.id)).map((row) => ({
|
|
1254
|
+
id: row.id,
|
|
1255
|
+
promptId: row.promptId,
|
|
1256
|
+
previousVersionId: "",
|
|
1257
|
+
newVersionId: row.versionId,
|
|
1258
|
+
previousPerformance: emptyMetrics(row.versionId),
|
|
1259
|
+
expectedPerformance: emptyMetrics(row.versionId),
|
|
1260
|
+
activatedAt: row.activatedAt,
|
|
1261
|
+
reason: row.reason ?? "",
|
|
1262
|
+
confidence: 0,
|
|
1263
|
+
monitoringStatus: "unknown"
|
|
1264
|
+
}));
|
|
1265
|
+
}
|
|
1266
|
+
} catch {
|
|
1267
|
+
}
|
|
1268
|
+
return [...inMemory, ...persisted].sort((a, b) => b.activatedAt.localeCompare(a.activatedAt));
|
|
1269
|
+
}
|
|
1270
|
+
/**
|
|
1271
|
+
* Close resources
|
|
1272
|
+
*/
|
|
1273
|
+
close() {
|
|
1274
|
+
this.manager.close();
|
|
1275
|
+
this.evaluator.close();
|
|
1276
|
+
this.optimizer.close();
|
|
1277
|
+
}
|
|
1278
|
+
// Private helper methods
|
|
1279
|
+
calculateScore(metrics) {
|
|
1280
|
+
const successScore = metrics.successRate * 100 * 0.35;
|
|
1281
|
+
const completenessScore = metrics.avgCompleteness * 0.35;
|
|
1282
|
+
const durationScore = Math.max(0, 1 - metrics.avgDuration / 300) * 100 * 0.15;
|
|
1283
|
+
const deliveryScore = metrics.avgCodeDeliveryRate * 100 * 0.15;
|
|
1284
|
+
return successScore + completenessScore + durationScore + deliveryScore;
|
|
1285
|
+
}
|
|
1286
|
+
async runSafetyChecks(currentPerf, candidatePerf, improvementPercentage) {
|
|
1287
|
+
const checks = [];
|
|
1288
|
+
checks.push({
|
|
1289
|
+
check: "sample_size",
|
|
1290
|
+
passed: candidatePerf.sampleSize >= this.options.minSampleSize,
|
|
1291
|
+
message: `Candidate has ${candidatePerf.sampleSize} samples (min: ${this.options.minSampleSize})`
|
|
1292
|
+
});
|
|
1293
|
+
const successRateChange = (candidatePerf.successRate - currentPerf.successRate) * 100;
|
|
1294
|
+
checks.push({
|
|
1295
|
+
check: "success_rate",
|
|
1296
|
+
passed: successRateChange >= -5,
|
|
1297
|
+
// Allow up to 5% decrease
|
|
1298
|
+
message: `Success rate change: ${successRateChange.toFixed(1)}%`
|
|
1299
|
+
});
|
|
1300
|
+
const completenessChange = candidatePerf.avgCompleteness - currentPerf.avgCompleteness;
|
|
1301
|
+
checks.push({
|
|
1302
|
+
check: "completeness",
|
|
1303
|
+
passed: completenessChange >= -10,
|
|
1304
|
+
// Allow up to 10 point decrease
|
|
1305
|
+
message: `Completeness change: ${completenessChange.toFixed(1)} points`
|
|
1306
|
+
});
|
|
1307
|
+
checks.push({
|
|
1308
|
+
check: "improvement_threshold",
|
|
1309
|
+
passed: improvementPercentage >= this.options.significanceThreshold,
|
|
1310
|
+
message: `Improvement: ${improvementPercentage.toFixed(1)}% (threshold: ${this.options.significanceThreshold}%)`
|
|
1311
|
+
});
|
|
1312
|
+
return checks;
|
|
1313
|
+
}
|
|
1314
|
+
calculateStatisticalSignificance(currentPerf, candidatePerf) {
|
|
1315
|
+
const n1 = currentPerf.sampleSize;
|
|
1316
|
+
const n2 = candidatePerf.sampleSize;
|
|
1317
|
+
const sampleFactor = Math.min((n1 + n2) / 40, 1);
|
|
1318
|
+
const avgStdDev = (currentPerf.stdDevCompleteness + candidatePerf.stdDevCompleteness) / 2;
|
|
1319
|
+
const variabilityFactor = Math.max(0, 1 - avgStdDev / 50);
|
|
1320
|
+
return (sampleFactor + variabilityFactor) / 2;
|
|
1321
|
+
}
|
|
1322
|
+
calculateConfidence(sampleSize, statisticalSignificance, improvementPercentage) {
|
|
1323
|
+
const sampleFactor = Math.min(sampleSize / 25, 1) * 0.4;
|
|
1324
|
+
const significanceFactor = statisticalSignificance * 0.3;
|
|
1325
|
+
const improvementFactor = Math.min(improvementPercentage / 30, 1) * 0.3;
|
|
1326
|
+
return sampleFactor + significanceFactor + improvementFactor;
|
|
1327
|
+
}
|
|
1328
|
+
generateActivationId() {
|
|
1329
|
+
return `act-${Date.now()}-${randomBytes(8).toString("hex")}`;
|
|
1330
|
+
}
|
|
1331
|
+
};
|
|
1332
|
+
|
|
1333
|
+
// src/prompts/lineage-tracker.ts
|
|
1334
|
+
import fs from "fs";
|
|
1335
|
+
import path from "path";
|
|
1336
|
+
import Database from "better-sqlite3";
|
|
1337
|
+
var LineageTracker = class {
|
|
1338
|
+
poeticDir;
|
|
1339
|
+
promptsDir;
|
|
1340
|
+
dbDir;
|
|
1341
|
+
db;
|
|
1342
|
+
debug;
|
|
1343
|
+
constructor(options = {}) {
|
|
1344
|
+
this.poeticDir = resolvePoeticDirFromContext(options.poeticDir);
|
|
1345
|
+
this.promptsDir = path.join(this.poeticDir, "prompts");
|
|
1346
|
+
this.dbDir = path.join(this.promptsDir, "db");
|
|
1347
|
+
this.debug = options.debug || false;
|
|
1348
|
+
this.ensureDirectories();
|
|
1349
|
+
const dbPath = path.join(this.dbDir, "prompts.db");
|
|
1350
|
+
this.db = new Database(dbPath);
|
|
1351
|
+
this.db.defaultSafeIntegers(false);
|
|
1352
|
+
this.initSchema();
|
|
1353
|
+
if (this.debug) {
|
|
1354
|
+
console.log("[LineageTracker] Initialized with database at:", dbPath);
|
|
1355
|
+
}
|
|
1356
|
+
}
|
|
1357
|
+
/**
|
|
1358
|
+
* Ensure all required directories exist
|
|
1359
|
+
*/
|
|
1360
|
+
ensureDirectories() {
|
|
1361
|
+
[this.promptsDir, this.dbDir].forEach((dir) => {
|
|
1362
|
+
if (!fs.existsSync(dir)) {
|
|
1363
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
1364
|
+
}
|
|
1365
|
+
});
|
|
1366
|
+
}
|
|
1367
|
+
/**
|
|
1368
|
+
* Initialize database schema for lineage tracking
|
|
1369
|
+
*/
|
|
1370
|
+
initSchema() {
|
|
1371
|
+
this.db.exec(`
|
|
1372
|
+
CREATE TABLE IF NOT EXISTS prompt_lineage (
|
|
1373
|
+
child_id TEXT PRIMARY KEY,
|
|
1374
|
+
parent_id TEXT,
|
|
1375
|
+
prompt_id TEXT NOT NULL,
|
|
1376
|
+
cycle_id TEXT,
|
|
1377
|
+
mutations TEXT NOT NULL,
|
|
1378
|
+
depth INTEGER DEFAULT 0,
|
|
1379
|
+
created_at TEXT NOT NULL,
|
|
1380
|
+
FOREIGN KEY (child_id) REFERENCES prompt_versions(id) ON DELETE CASCADE,
|
|
1381
|
+
FOREIGN KEY (parent_id) REFERENCES prompt_versions(id) ON DELETE SET NULL
|
|
1382
|
+
);
|
|
1383
|
+
|
|
1384
|
+
CREATE INDEX IF NOT EXISTS idx_lineage_prompt ON prompt_lineage(prompt_id);
|
|
1385
|
+
CREATE INDEX IF NOT EXISTS idx_lineage_parent ON prompt_lineage(parent_id);
|
|
1386
|
+
CREATE INDEX IF NOT EXISTS idx_lineage_cycle ON prompt_lineage(cycle_id);
|
|
1387
|
+
CREATE INDEX IF NOT EXISTS idx_lineage_depth ON prompt_lineage(depth);
|
|
1388
|
+
`);
|
|
1389
|
+
}
|
|
1390
|
+
/**
|
|
1391
|
+
* Track a mutation from parent to child version
|
|
1392
|
+
*
|
|
1393
|
+
* Creates a lineage entry recording the parent-child relationship,
|
|
1394
|
+
* mutations applied, and optimization cycle.
|
|
1395
|
+
*
|
|
1396
|
+
* @param parentId - Parent version ID (null for root versions)
|
|
1397
|
+
* @param childId - Child version ID
|
|
1398
|
+
* @param mutations - Array of mutations applied
|
|
1399
|
+
* @param cycleId - Optimization cycle identifier (optional)
|
|
1400
|
+
* @throws Error if child already has lineage recorded
|
|
1401
|
+
*
|
|
1402
|
+
* @example
|
|
1403
|
+
* ```typescript
|
|
1404
|
+
* await tracker.trackMutation('v1.0.0', 'v1.1.0', [
|
|
1405
|
+
* {
|
|
1406
|
+
* type: 'add_context',
|
|
1407
|
+
* description: 'Added TypeScript specific examples',
|
|
1408
|
+
* impact: 0.7
|
|
1409
|
+
* }
|
|
1410
|
+
* ], 'optimization-cycle-1');
|
|
1411
|
+
* ```
|
|
1412
|
+
*/
|
|
1413
|
+
async trackMutation(parentId, childId, mutations, cycleId) {
|
|
1414
|
+
const existingStmt = this.db.prepare("SELECT child_id FROM prompt_lineage WHERE child_id = ?");
|
|
1415
|
+
const existing = existingStmt.get(childId);
|
|
1416
|
+
if (existing) {
|
|
1417
|
+
throw new Error(`Lineage already exists for version "${childId}"`);
|
|
1418
|
+
}
|
|
1419
|
+
let depth = 0;
|
|
1420
|
+
let promptId;
|
|
1421
|
+
if (parentId) {
|
|
1422
|
+
const parentLineage = this.getLineageEntry(parentId);
|
|
1423
|
+
if (parentLineage) {
|
|
1424
|
+
depth = parentLineage.depth + 1;
|
|
1425
|
+
promptId = parentLineage.promptId;
|
|
1426
|
+
} else {
|
|
1427
|
+
const versionStmt = this.db.prepare("SELECT prompt_id FROM prompt_versions WHERE id = ?");
|
|
1428
|
+
const parentVersion = versionStmt.get(parentId);
|
|
1429
|
+
if (!parentVersion) {
|
|
1430
|
+
throw new Error(`Parent version "${parentId}" not found`);
|
|
1431
|
+
}
|
|
1432
|
+
promptId = parentVersion.prompt_id;
|
|
1433
|
+
depth = 1;
|
|
1434
|
+
}
|
|
1435
|
+
} else {
|
|
1436
|
+
const versionStmt = this.db.prepare("SELECT prompt_id FROM prompt_versions WHERE id = ?");
|
|
1437
|
+
const childVersion = versionStmt.get(childId);
|
|
1438
|
+
if (!childVersion) {
|
|
1439
|
+
throw new Error(`Child version "${childId}" not found`);
|
|
1440
|
+
}
|
|
1441
|
+
promptId = childVersion.prompt_id;
|
|
1442
|
+
depth = 0;
|
|
1443
|
+
}
|
|
1444
|
+
const now = (/* @__PURE__ */ new Date()).toISOString();
|
|
1445
|
+
const mutationsJson = JSON.stringify(mutations);
|
|
1446
|
+
const stmt = this.db.prepare(`
|
|
1447
|
+
INSERT INTO prompt_lineage (
|
|
1448
|
+
child_id, parent_id, prompt_id, cycle_id, mutations, depth, created_at
|
|
1449
|
+
)
|
|
1450
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
1451
|
+
`);
|
|
1452
|
+
stmt.run(childId, parentId, promptId, cycleId || null, mutationsJson, depth, now);
|
|
1453
|
+
if (this.debug) {
|
|
1454
|
+
console.log("[LineageTracker] Tracked mutation:", {
|
|
1455
|
+
parentId,
|
|
1456
|
+
childId,
|
|
1457
|
+
promptId,
|
|
1458
|
+
depth,
|
|
1459
|
+
mutations: mutations.length
|
|
1460
|
+
});
|
|
1461
|
+
}
|
|
1462
|
+
return {
|
|
1463
|
+
childId,
|
|
1464
|
+
parentId,
|
|
1465
|
+
promptId,
|
|
1466
|
+
cycleId: cycleId || null,
|
|
1467
|
+
mutations,
|
|
1468
|
+
depth,
|
|
1469
|
+
createdAt: now
|
|
1470
|
+
};
|
|
1471
|
+
}
|
|
1472
|
+
/**
|
|
1473
|
+
* Get lineage for a prompt, ordered from newest to oldest
|
|
1474
|
+
*
|
|
1475
|
+
* Returns the full version history in chronological order (newest first),
|
|
1476
|
+
* with each version's metadata from the prompt_versions table.
|
|
1477
|
+
*
|
|
1478
|
+
* @param promptId - Prompt ID to get lineage for
|
|
1479
|
+
* @returns Array of versions ordered newest to oldest
|
|
1480
|
+
*
|
|
1481
|
+
* @example
|
|
1482
|
+
* ```typescript
|
|
1483
|
+
* const lineage = await tracker.getLineage('my-prompt');
|
|
1484
|
+
* console.log('Latest version:', lineage[0]);
|
|
1485
|
+
* console.log('Original version:', lineage[lineage.length - 1]);
|
|
1486
|
+
* ```
|
|
1487
|
+
*/
|
|
1488
|
+
async getLineage(promptId) {
|
|
1489
|
+
const stmt = this.db.prepare(`
|
|
1490
|
+
SELECT * FROM prompt_versions
|
|
1491
|
+
WHERE prompt_id = ?
|
|
1492
|
+
ORDER BY created_at DESC
|
|
1493
|
+
`);
|
|
1494
|
+
const rows = stmt.all(promptId);
|
|
1495
|
+
return rows.map((row) => this.rowToVersion(row, promptId));
|
|
1496
|
+
}
|
|
1497
|
+
/**
|
|
1498
|
+
* Get evolution tree for a prompt
|
|
1499
|
+
*
|
|
1500
|
+
* Builds a complete tree structure showing all versions and their
|
|
1501
|
+
* parent-child relationships, with mutation information.
|
|
1502
|
+
*
|
|
1503
|
+
* @param promptId - Prompt ID to build tree for
|
|
1504
|
+
* @returns Evolution tree structure
|
|
1505
|
+
* @throws Error if no versions exist for the prompt
|
|
1506
|
+
*
|
|
1507
|
+
* @example
|
|
1508
|
+
* ```typescript
|
|
1509
|
+
* const tree = await tracker.getEvolutionTree('my-prompt');
|
|
1510
|
+
* console.log('Root version:', tree.rootId);
|
|
1511
|
+
* console.log('Total versions:', tree.totalVersions);
|
|
1512
|
+
* console.log('Max depth:', tree.maxDepth);
|
|
1513
|
+
*
|
|
1514
|
+
* // Walk the tree
|
|
1515
|
+
* const root = tree.nodes.get(tree.rootId);
|
|
1516
|
+
* root?.children.forEach(childId => {
|
|
1517
|
+
* const child = tree.nodes.get(childId);
|
|
1518
|
+
* console.log('Child mutations:', child?.mutations);
|
|
1519
|
+
* });
|
|
1520
|
+
* ```
|
|
1521
|
+
*/
|
|
1522
|
+
async getEvolutionTree(promptId) {
|
|
1523
|
+
const lineageStmt = this.db.prepare(`
|
|
1524
|
+
SELECT * FROM prompt_lineage
|
|
1525
|
+
WHERE prompt_id = ?
|
|
1526
|
+
ORDER BY depth ASC, created_at ASC
|
|
1527
|
+
`);
|
|
1528
|
+
const lineageRows = lineageStmt.all(promptId);
|
|
1529
|
+
if (lineageRows.length === 0) {
|
|
1530
|
+
throw new Error(`No lineage found for prompt "${promptId}"`);
|
|
1531
|
+
}
|
|
1532
|
+
const versionsStmt = this.db.prepare(`
|
|
1533
|
+
SELECT * FROM prompt_versions
|
|
1534
|
+
WHERE prompt_id = ?
|
|
1535
|
+
`);
|
|
1536
|
+
const versionRows = versionsStmt.all(promptId);
|
|
1537
|
+
const versionMap = /* @__PURE__ */ new Map();
|
|
1538
|
+
let activeVersionId = null;
|
|
1539
|
+
versionRows.forEach((row) => {
|
|
1540
|
+
const version = this.rowToVersion(row, promptId);
|
|
1541
|
+
versionMap.set(row.id, version);
|
|
1542
|
+
if (row.is_active === 1) {
|
|
1543
|
+
activeVersionId = row.id;
|
|
1544
|
+
}
|
|
1545
|
+
});
|
|
1546
|
+
const nodes = /* @__PURE__ */ new Map();
|
|
1547
|
+
let rootId = null;
|
|
1548
|
+
let maxDepth = 0;
|
|
1549
|
+
lineageRows.forEach((row) => {
|
|
1550
|
+
const mutations = parseJsonWithContext(
|
|
1551
|
+
row.mutations,
|
|
1552
|
+
"prompt mutations for evolution tree"
|
|
1553
|
+
);
|
|
1554
|
+
const version = versionMap.get(row.child_id) || null;
|
|
1555
|
+
const isActive = row.child_id === activeVersionId;
|
|
1556
|
+
const node = {
|
|
1557
|
+
versionId: row.child_id,
|
|
1558
|
+
version,
|
|
1559
|
+
parentId: row.parent_id,
|
|
1560
|
+
children: [],
|
|
1561
|
+
mutations,
|
|
1562
|
+
cycleId: row.cycle_id,
|
|
1563
|
+
depth: row.depth,
|
|
1564
|
+
isActive,
|
|
1565
|
+
createdAt: row.created_at
|
|
1566
|
+
};
|
|
1567
|
+
nodes.set(row.child_id, node);
|
|
1568
|
+
if (row.depth === 0 && !row.parent_id) {
|
|
1569
|
+
rootId = row.child_id;
|
|
1570
|
+
}
|
|
1571
|
+
maxDepth = Math.max(maxDepth, row.depth);
|
|
1572
|
+
});
|
|
1573
|
+
nodes.forEach((node) => {
|
|
1574
|
+
if (node.parentId) {
|
|
1575
|
+
const parent = nodes.get(node.parentId);
|
|
1576
|
+
if (parent) {
|
|
1577
|
+
parent.children.push(node.versionId);
|
|
1578
|
+
}
|
|
1579
|
+
}
|
|
1580
|
+
});
|
|
1581
|
+
if (!rootId) {
|
|
1582
|
+
throw new Error(`No root version found for prompt "${promptId}"`);
|
|
1583
|
+
}
|
|
1584
|
+
return {
|
|
1585
|
+
rootId,
|
|
1586
|
+
nodes,
|
|
1587
|
+
maxDepth,
|
|
1588
|
+
totalVersions: nodes.size,
|
|
1589
|
+
activeVersionId
|
|
1590
|
+
};
|
|
1591
|
+
}
|
|
1592
|
+
/**
|
|
1593
|
+
* Rollback prompt to a previous version
|
|
1594
|
+
*
|
|
1595
|
+
* Activates the target version, deactivating the current active version.
|
|
1596
|
+
* Does not delete any versions - all history is preserved.
|
|
1597
|
+
*
|
|
1598
|
+
* @param promptId - Prompt ID to rollback
|
|
1599
|
+
* @param targetVersionId - Version ID to activate
|
|
1600
|
+
* @throws Error if version doesn't exist or doesn't belong to prompt
|
|
1601
|
+
*
|
|
1602
|
+
* @example
|
|
1603
|
+
* ```typescript
|
|
1604
|
+
* // Rollback to previous version
|
|
1605
|
+
* await tracker.rollback('my-prompt', 'v1.0.0');
|
|
1606
|
+
* ```
|
|
1607
|
+
*/
|
|
1608
|
+
async rollback(promptId, targetVersionId) {
|
|
1609
|
+
const versionStmt = this.db.prepare(
|
|
1610
|
+
"SELECT prompt_id, is_active FROM prompt_versions WHERE id = ?"
|
|
1611
|
+
);
|
|
1612
|
+
const version = versionStmt.get(targetVersionId);
|
|
1613
|
+
if (!version) {
|
|
1614
|
+
throw new Error(`Version "${targetVersionId}" not found`);
|
|
1615
|
+
}
|
|
1616
|
+
if (version.prompt_id !== promptId) {
|
|
1617
|
+
throw new Error(`Version "${targetVersionId}" does not belong to prompt "${promptId}"`);
|
|
1618
|
+
}
|
|
1619
|
+
if (version.is_active === 1) {
|
|
1620
|
+
if (this.debug) {
|
|
1621
|
+
console.log("[LineageTracker] Version already active, no rollback needed");
|
|
1622
|
+
}
|
|
1623
|
+
return;
|
|
1624
|
+
}
|
|
1625
|
+
const now = (/* @__PURE__ */ new Date()).toISOString();
|
|
1626
|
+
const deactivateStmt = this.db.prepare(`
|
|
1627
|
+
UPDATE prompt_versions
|
|
1628
|
+
SET is_active = 0, deactivated_at = ?
|
|
1629
|
+
WHERE prompt_id = ? AND is_active = 1
|
|
1630
|
+
`);
|
|
1631
|
+
const activateStmt = this.db.prepare(`
|
|
1632
|
+
UPDATE prompt_versions
|
|
1633
|
+
SET is_active = 1, activated_at = ?
|
|
1634
|
+
WHERE id = ?
|
|
1635
|
+
`);
|
|
1636
|
+
const updatePromptStmt = this.db.prepare(`
|
|
1637
|
+
UPDATE prompts
|
|
1638
|
+
SET current_version = ?, updated_at = ?
|
|
1639
|
+
WHERE id = ?
|
|
1640
|
+
`);
|
|
1641
|
+
const transaction = this.db.transaction(() => {
|
|
1642
|
+
deactivateStmt.run(now, promptId);
|
|
1643
|
+
activateStmt.run(now, targetVersionId);
|
|
1644
|
+
updatePromptStmt.run(targetVersionId, now, promptId);
|
|
1645
|
+
});
|
|
1646
|
+
transaction();
|
|
1647
|
+
if (this.debug) {
|
|
1648
|
+
console.log("[LineageTracker] Rolled back prompt to version:", targetVersionId);
|
|
1649
|
+
}
|
|
1650
|
+
}
|
|
1651
|
+
/**
|
|
1652
|
+
* Get lineage entry for a specific version
|
|
1653
|
+
*
|
|
1654
|
+
* @param versionId - Version ID to get lineage for
|
|
1655
|
+
* @returns Lineage entry or null if not found
|
|
1656
|
+
*/
|
|
1657
|
+
getLineageEntry(versionId) {
|
|
1658
|
+
const stmt = this.db.prepare("SELECT * FROM prompt_lineage WHERE child_id = ?");
|
|
1659
|
+
const row = stmt.get(versionId);
|
|
1660
|
+
if (!row) {
|
|
1661
|
+
return null;
|
|
1662
|
+
}
|
|
1663
|
+
return this.rowToLineageEntry(row);
|
|
1664
|
+
}
|
|
1665
|
+
/**
|
|
1666
|
+
* Get all children of a version
|
|
1667
|
+
*
|
|
1668
|
+
* @param parentId - Parent version ID
|
|
1669
|
+
* @returns Array of child lineage entries
|
|
1670
|
+
*/
|
|
1671
|
+
getChildren(parentId) {
|
|
1672
|
+
const stmt = this.db.prepare(`
|
|
1673
|
+
SELECT * FROM prompt_lineage
|
|
1674
|
+
WHERE parent_id = ?
|
|
1675
|
+
ORDER BY created_at ASC
|
|
1676
|
+
`);
|
|
1677
|
+
const rows = stmt.all(parentId);
|
|
1678
|
+
return rows.map((row) => this.rowToLineageEntry(row));
|
|
1679
|
+
}
|
|
1680
|
+
/**
|
|
1681
|
+
* Get all versions in an optimization cycle
|
|
1682
|
+
*
|
|
1683
|
+
* @param cycleId - Optimization cycle ID
|
|
1684
|
+
* @returns Array of lineage entries in the cycle
|
|
1685
|
+
*/
|
|
1686
|
+
getCycleVersions(cycleId) {
|
|
1687
|
+
const stmt = this.db.prepare(`
|
|
1688
|
+
SELECT * FROM prompt_lineage
|
|
1689
|
+
WHERE cycle_id = ?
|
|
1690
|
+
ORDER BY depth ASC, created_at ASC
|
|
1691
|
+
`);
|
|
1692
|
+
const rows = stmt.all(cycleId);
|
|
1693
|
+
return rows.map((row) => this.rowToLineageEntry(row));
|
|
1694
|
+
}
|
|
1695
|
+
/**
|
|
1696
|
+
* Get mutation statistics for a prompt
|
|
1697
|
+
*
|
|
1698
|
+
* Analyzes all mutations across the lineage to provide insights
|
|
1699
|
+
* into evolution patterns.
|
|
1700
|
+
*
|
|
1701
|
+
* @param promptId - Prompt ID to analyze
|
|
1702
|
+
* @returns Mutation statistics
|
|
1703
|
+
*/
|
|
1704
|
+
getMutationStats(promptId) {
|
|
1705
|
+
const stmt = this.db.prepare("SELECT mutations FROM prompt_lineage WHERE prompt_id = ?");
|
|
1706
|
+
const rows = stmt.all(promptId);
|
|
1707
|
+
let totalMutations = 0;
|
|
1708
|
+
const mutationsByType = {};
|
|
1709
|
+
let totalImpact = 0;
|
|
1710
|
+
let impactCount = 0;
|
|
1711
|
+
rows.forEach((row) => {
|
|
1712
|
+
const mutations = parseJsonWithContext(
|
|
1713
|
+
row.mutations,
|
|
1714
|
+
"prompt mutations for stats"
|
|
1715
|
+
);
|
|
1716
|
+
totalMutations += mutations.length;
|
|
1717
|
+
mutations.forEach((mutation) => {
|
|
1718
|
+
mutationsByType[mutation.type] = (mutationsByType[mutation.type] || 0) + 1;
|
|
1719
|
+
if (mutation.impact !== void 0) {
|
|
1720
|
+
totalImpact += mutation.impact;
|
|
1721
|
+
impactCount++;
|
|
1722
|
+
}
|
|
1723
|
+
});
|
|
1724
|
+
});
|
|
1725
|
+
return {
|
|
1726
|
+
totalMutations,
|
|
1727
|
+
mutationsByType,
|
|
1728
|
+
avgMutationsPerVersion: rows.length > 0 ? totalMutations / rows.length : 0,
|
|
1729
|
+
avgImpact: impactCount > 0 ? totalImpact / impactCount : 0
|
|
1730
|
+
};
|
|
1731
|
+
}
|
|
1732
|
+
/**
|
|
1733
|
+
* Close database connection
|
|
1734
|
+
*
|
|
1735
|
+
* Should be called when done using the tracker to release resources.
|
|
1736
|
+
*
|
|
1737
|
+
* @example
|
|
1738
|
+
* ```typescript
|
|
1739
|
+
* const tracker = new LineageTracker();
|
|
1740
|
+
* // ... use tracker ...
|
|
1741
|
+
* tracker.close();
|
|
1742
|
+
* ```
|
|
1743
|
+
*/
|
|
1744
|
+
close() {
|
|
1745
|
+
this.db.close();
|
|
1746
|
+
if (this.debug) {
|
|
1747
|
+
console.log("[LineageTracker] Closed database connection");
|
|
1748
|
+
}
|
|
1749
|
+
}
|
|
1750
|
+
// Private helper methods
|
|
1751
|
+
/**
|
|
1752
|
+
* Convert database row to PromptVersion
|
|
1753
|
+
*/
|
|
1754
|
+
rowToVersion(row, promptId) {
|
|
1755
|
+
return {
|
|
1756
|
+
id: row.id,
|
|
1757
|
+
promptId,
|
|
1758
|
+
content: "",
|
|
1759
|
+
// Empty - fetch from PromptStore if needed
|
|
1760
|
+
contentHash: row.content_hash,
|
|
1761
|
+
changelog: row.changelog || void 0,
|
|
1762
|
+
author: row.author || void 0,
|
|
1763
|
+
isActive: row.is_active === 1,
|
|
1764
|
+
isGenerated: row.is_generated === 1,
|
|
1765
|
+
generationModel: row.generation_model || void 0,
|
|
1766
|
+
generationPrompt: row.generation_prompt || void 0,
|
|
1767
|
+
createdAt: row.created_at,
|
|
1768
|
+
activatedAt: row.activated_at || void 0,
|
|
1769
|
+
deactivatedAt: row.deactivated_at || void 0
|
|
1770
|
+
};
|
|
1771
|
+
}
|
|
1772
|
+
/**
|
|
1773
|
+
* Convert database row to LineageEntry
|
|
1774
|
+
*/
|
|
1775
|
+
rowToLineageEntry(row) {
|
|
1776
|
+
const mutations = parseJsonWithContext(
|
|
1777
|
+
row.mutations,
|
|
1778
|
+
"prompt mutations for lineage entry"
|
|
1779
|
+
);
|
|
1780
|
+
return {
|
|
1781
|
+
childId: row.child_id,
|
|
1782
|
+
parentId: row.parent_id,
|
|
1783
|
+
promptId: row.prompt_id,
|
|
1784
|
+
cycleId: row.cycle_id,
|
|
1785
|
+
mutations,
|
|
1786
|
+
depth: row.depth,
|
|
1787
|
+
createdAt: row.created_at
|
|
1788
|
+
};
|
|
1789
|
+
}
|
|
1790
|
+
};
|
|
1791
|
+
|
|
1792
|
+
// src/prompts/provider-tracker.ts
|
|
1793
|
+
import fs2 from "fs";
|
|
1794
|
+
import path2 from "path";
|
|
1795
|
+
import Database2 from "better-sqlite3";
|
|
1796
|
+
|
|
1797
|
+
// src/core/evaluation/scoring-telemetry-event.ts
|
|
1798
|
+
function createTelemetryEvent(eventType, data, metadata) {
|
|
1799
|
+
return {
|
|
1800
|
+
eventType,
|
|
1801
|
+
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1802
|
+
data,
|
|
1803
|
+
metadata
|
|
1804
|
+
};
|
|
1805
|
+
}
|
|
1806
|
+
|
|
1807
|
+
// src/core/evaluation/unified-scoring.ts
|
|
1808
|
+
function clamp01(x) {
|
|
1809
|
+
if (!Number.isFinite(x)) return 0;
|
|
1810
|
+
return Math.max(0, Math.min(1, x));
|
|
1811
|
+
}
|
|
1812
|
+
function clamp100(x) {
|
|
1813
|
+
if (!Number.isFinite(x)) return 0;
|
|
1814
|
+
return Math.max(0, Math.min(100, x));
|
|
1815
|
+
}
|
|
1816
|
+
function aggregateScores(scores, options = {}) {
|
|
1817
|
+
const cfg = options.criteriaConfig ?? DEFAULT_EVALUATION_CRITERIA;
|
|
1818
|
+
const active = getActiveCriteria(cfg);
|
|
1819
|
+
const reasons = [];
|
|
1820
|
+
let disqualified = false;
|
|
1821
|
+
for (const [key, criterion] of Object.entries(active)) {
|
|
1822
|
+
const s = clamp100(scores[key] ?? 0);
|
|
1823
|
+
if (s < criterion.minThreshold) {
|
|
1824
|
+
reasons.push(`${key} below threshold: ${s.toFixed(1)} < ${criterion.minThreshold}`);
|
|
1825
|
+
if (options.requireAllThresholds ?? cfg.defaultStrategy.requireAllThresholds) {
|
|
1826
|
+
disqualified = true;
|
|
1827
|
+
}
|
|
1828
|
+
}
|
|
1829
|
+
}
|
|
1830
|
+
const algo = options.baseAlgorithm ?? "harmonic-mean";
|
|
1831
|
+
const entries = Object.entries(active).map(([key, c]) => ({
|
|
1832
|
+
key,
|
|
1833
|
+
weight: clamp01(c.weight),
|
|
1834
|
+
value: clamp01((scores[key] ?? 0) / 100)
|
|
1835
|
+
}));
|
|
1836
|
+
if (isTelemetryDebug()) {
|
|
1837
|
+
try {
|
|
1838
|
+
const event = createTelemetryEvent("scoring.aggregate_entries", {
|
|
1839
|
+
entries: entries.map((e) => ({ key: e.key, weight: e.weight, value: e.value * 100 })),
|
|
1840
|
+
algorithm: algo
|
|
1841
|
+
});
|
|
1842
|
+
if (typeof process !== "undefined" && process.stderr) {
|
|
1843
|
+
process.stderr.write(`${JSON.stringify(event)}
|
|
1844
|
+
`);
|
|
1845
|
+
}
|
|
1846
|
+
} catch {
|
|
1847
|
+
}
|
|
1848
|
+
}
|
|
1849
|
+
const weightSum = entries.reduce((s, e) => s + e.weight, 0) || 1;
|
|
1850
|
+
const normalizedEntries = entries.map((entry) => ({
|
|
1851
|
+
...entry,
|
|
1852
|
+
weight: entry.weight / weightSum
|
|
1853
|
+
}));
|
|
1854
|
+
let value01 = 0;
|
|
1855
|
+
if (algo === "weighted-sum") {
|
|
1856
|
+
value01 = normalizedEntries.reduce((s, e) => s + e.value * e.weight, 0);
|
|
1857
|
+
} else {
|
|
1858
|
+
const eps = 1e-6;
|
|
1859
|
+
const numerator = normalizedEntries.reduce((s, e) => s + e.weight, 0);
|
|
1860
|
+
const denominator = normalizedEntries.reduce(
|
|
1861
|
+
(s, e) => s + e.weight / Math.max(e.value, eps),
|
|
1862
|
+
0
|
|
1863
|
+
);
|
|
1864
|
+
value01 = numerator / Math.max(denominator, eps);
|
|
1865
|
+
}
|
|
1866
|
+
const score = clamp100(value01 * 100);
|
|
1867
|
+
const minOverall = options.minOverallScore ?? cfg.defaultStrategy.minOverallScore;
|
|
1868
|
+
if (score < minOverall) {
|
|
1869
|
+
reasons.push(`overall score below minimum: ${score.toFixed(1)} < ${minOverall}`);
|
|
1870
|
+
disqualified = true;
|
|
1871
|
+
}
|
|
1872
|
+
return { score, disqualified, reasons };
|
|
1873
|
+
}
|
|
1874
|
+
function buildCriterionScores(partial, criteriaConfig = DEFAULT_EVALUATION_CRITERIA, fallbackValue = JUDGE_DEFAULTS.CRITERION_FALLBACK) {
|
|
1875
|
+
const active = getActiveCriteria(criteriaConfig);
|
|
1876
|
+
const out = {};
|
|
1877
|
+
const defaulted = [];
|
|
1878
|
+
for (const key of Object.keys(active)) {
|
|
1879
|
+
const v = partial[key];
|
|
1880
|
+
if (typeof v === "number" && Number.isFinite(v)) {
|
|
1881
|
+
out[key] = clamp100(v);
|
|
1882
|
+
} else {
|
|
1883
|
+
out[key] = clamp100(fallbackValue);
|
|
1884
|
+
defaulted.push(key);
|
|
1885
|
+
}
|
|
1886
|
+
}
|
|
1887
|
+
if (defaulted.length > 0 && isTelemetryDebug()) {
|
|
1888
|
+
try {
|
|
1889
|
+
const event = createTelemetryEvent("scoring.build_defaults", {
|
|
1890
|
+
fallbackValue,
|
|
1891
|
+
defaulted
|
|
1892
|
+
});
|
|
1893
|
+
if (typeof process !== "undefined" && process.stderr) {
|
|
1894
|
+
process.stderr.write(`${JSON.stringify(event)}
|
|
1895
|
+
`);
|
|
1896
|
+
}
|
|
1897
|
+
} catch {
|
|
1898
|
+
}
|
|
1899
|
+
}
|
|
1900
|
+
return out;
|
|
1901
|
+
}
|
|
1902
|
+
|
|
1903
|
+
// src/prompts/provider-tracker.ts
|
|
1904
|
+
var ProviderTracker = class {
|
|
1905
|
+
poeticDir;
|
|
1906
|
+
dbPath;
|
|
1907
|
+
db;
|
|
1908
|
+
options;
|
|
1909
|
+
// Valid providers, derived from the registry so newly registered or
|
|
1910
|
+
// user-defined providers are tracked rather than silently rejected [B7].
|
|
1911
|
+
// Resolved once in the constructor from ProviderRegistry.getAllProviderIds().
|
|
1912
|
+
validProviders;
|
|
1913
|
+
validProviderSet;
|
|
1914
|
+
constructor(options = {}) {
|
|
1915
|
+
this.poeticDir = resolvePoeticDirFromContext(options.poeticDir);
|
|
1916
|
+
const dbDir = path2.join(this.poeticDir, "prompts", "db");
|
|
1917
|
+
if (!fs2.existsSync(dbDir)) {
|
|
1918
|
+
fs2.mkdirSync(dbDir, { recursive: true });
|
|
1919
|
+
}
|
|
1920
|
+
this.dbPath = path2.join(dbDir, "provider-performance.db");
|
|
1921
|
+
this.db = new Database2(this.dbPath);
|
|
1922
|
+
let registeredProviders = [];
|
|
1923
|
+
try {
|
|
1924
|
+
registeredProviders = ProviderRegistry.getAllProviderIds();
|
|
1925
|
+
} catch {
|
|
1926
|
+
registeredProviders = [];
|
|
1927
|
+
}
|
|
1928
|
+
this.validProviders = registeredProviders;
|
|
1929
|
+
this.validProviderSet = new Set(registeredProviders);
|
|
1930
|
+
this.options = {
|
|
1931
|
+
poeticDir: this.poeticDir,
|
|
1932
|
+
minSampleSize: options.minSampleSize ?? 5,
|
|
1933
|
+
confidenceLevel: options.confidenceLevel ?? 0.95,
|
|
1934
|
+
explorationRate: options.explorationRate ?? 0.1,
|
|
1935
|
+
enablePrediction: options.enablePrediction ?? true
|
|
1936
|
+
};
|
|
1937
|
+
this.initSchema();
|
|
1938
|
+
}
|
|
1939
|
+
/**
|
|
1940
|
+
* Initialize database schema
|
|
1941
|
+
*/
|
|
1942
|
+
initSchema() {
|
|
1943
|
+
this.db.exec(`
|
|
1944
|
+
CREATE TABLE IF NOT EXISTS provider_executions (
|
|
1945
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
1946
|
+
provider TEXT NOT NULL,
|
|
1947
|
+
prompt_id TEXT NOT NULL,
|
|
1948
|
+
version_id TEXT NOT NULL,
|
|
1949
|
+
variant TEXT,
|
|
1950
|
+
|
|
1951
|
+
-- Performance metrics
|
|
1952
|
+
success INTEGER NOT NULL,
|
|
1953
|
+
completeness REAL NOT NULL,
|
|
1954
|
+
duration REAL NOT NULL,
|
|
1955
|
+
tokens_used INTEGER,
|
|
1956
|
+
code_delivered INTEGER NOT NULL,
|
|
1957
|
+
|
|
1958
|
+
-- Quality metrics
|
|
1959
|
+
error_count INTEGER DEFAULT 0,
|
|
1960
|
+
warning_count INTEGER DEFAULT 0,
|
|
1961
|
+
tests_passed INTEGER,
|
|
1962
|
+
tests_failed INTEGER,
|
|
1963
|
+
|
|
1964
|
+
-- Context
|
|
1965
|
+
task_complexity TEXT,
|
|
1966
|
+
codebase_size TEXT,
|
|
1967
|
+
timestamp TEXT NOT NULL,
|
|
1968
|
+
|
|
1969
|
+
-- Metadata
|
|
1970
|
+
created_at TEXT DEFAULT CURRENT_TIMESTAMP
|
|
1971
|
+
);
|
|
1972
|
+
|
|
1973
|
+
CREATE INDEX IF NOT EXISTS idx_provider_prompt
|
|
1974
|
+
ON provider_executions(provider, prompt_id, version_id);
|
|
1975
|
+
CREATE INDEX IF NOT EXISTS idx_provider_timestamp
|
|
1976
|
+
ON provider_executions(provider, timestamp DESC);
|
|
1977
|
+
CREATE INDEX IF NOT EXISTS idx_prompt_provider
|
|
1978
|
+
ON provider_executions(prompt_id, provider);
|
|
1979
|
+
|
|
1980
|
+
-- Aggregated statistics cache for performance
|
|
1981
|
+
CREATE TABLE IF NOT EXISTS provider_stats_cache (
|
|
1982
|
+
provider TEXT NOT NULL,
|
|
1983
|
+
prompt_id TEXT NOT NULL,
|
|
1984
|
+
version_id TEXT NOT NULL,
|
|
1985
|
+
variant TEXT NOT NULL DEFAULT '',
|
|
1986
|
+
|
|
1987
|
+
-- Cached metrics
|
|
1988
|
+
total_executions INTEGER NOT NULL,
|
|
1989
|
+
success_rate REAL NOT NULL,
|
|
1990
|
+
avg_completeness REAL NOT NULL,
|
|
1991
|
+
avg_duration REAL NOT NULL,
|
|
1992
|
+
std_dev_completeness REAL NOT NULL,
|
|
1993
|
+
median_completeness REAL NOT NULL,
|
|
1994
|
+
|
|
1995
|
+
-- Trend analysis
|
|
1996
|
+
performance_trend TEXT NOT NULL,
|
|
1997
|
+
trend_strength REAL NOT NULL,
|
|
1998
|
+
|
|
1999
|
+
-- Metadata
|
|
2000
|
+
calculated_at TEXT NOT NULL,
|
|
2001
|
+
sample_size INTEGER NOT NULL,
|
|
2002
|
+
|
|
2003
|
+
PRIMARY KEY (provider, prompt_id, version_id, variant)
|
|
2004
|
+
);
|
|
2005
|
+
|
|
2006
|
+
-- Provider affinity scores for contextual recommendations
|
|
2007
|
+
CREATE TABLE IF NOT EXISTS provider_affinity (
|
|
2008
|
+
provider TEXT NOT NULL,
|
|
2009
|
+
prompt_id TEXT NOT NULL,
|
|
2010
|
+
context_hash TEXT NOT NULL,
|
|
2011
|
+
affinity_score REAL NOT NULL,
|
|
2012
|
+
context_factors TEXT NOT NULL,
|
|
2013
|
+
calculated_at TEXT NOT NULL,
|
|
2014
|
+
|
|
2015
|
+
PRIMARY KEY (provider, prompt_id, context_hash)
|
|
2016
|
+
);
|
|
2017
|
+
`);
|
|
2018
|
+
}
|
|
2019
|
+
/**
|
|
2020
|
+
* Track a provider execution
|
|
2021
|
+
*
|
|
2022
|
+
* Records performance metrics for a specific provider execution. This is the
|
|
2023
|
+
* primary method for feeding data into the analytics system.
|
|
2024
|
+
*
|
|
2025
|
+
* @param metrics Execution metrics to track
|
|
2026
|
+
* @throws Error if provider is invalid or metrics are incomplete
|
|
2027
|
+
*
|
|
2028
|
+
* @example
|
|
2029
|
+
* ```typescript
|
|
2030
|
+
* await tracker.trackProviderExecution({
|
|
2031
|
+
* provider: 'claude',
|
|
2032
|
+
* promptId: 'typescript-expert',
|
|
2033
|
+
* versionId: 'v1.2.0',
|
|
2034
|
+
* variant: 'conservative',
|
|
2035
|
+
* success: true,
|
|
2036
|
+
* completeness: 95,
|
|
2037
|
+
* duration: 45.2,
|
|
2038
|
+
* codeDelivered: true,
|
|
2039
|
+
* errorCount: 0,
|
|
2040
|
+
* timestamp: new Date().toISOString()
|
|
2041
|
+
* });
|
|
2042
|
+
* ```
|
|
2043
|
+
*/
|
|
2044
|
+
async trackProviderExecution(metrics) {
|
|
2045
|
+
try {
|
|
2046
|
+
if (!this.isValidProvider(metrics.provider)) {
|
|
2047
|
+
throw new Error(
|
|
2048
|
+
`Invalid provider: ${metrics.provider}. Must be one of: ${this.validProviders.join(", ")}`
|
|
2049
|
+
);
|
|
2050
|
+
}
|
|
2051
|
+
if (!metrics.promptId || !metrics.versionId) {
|
|
2052
|
+
throw new Error("promptId and versionId are required");
|
|
2053
|
+
}
|
|
2054
|
+
if (metrics.completeness < 0 || metrics.completeness > 100) {
|
|
2055
|
+
throw new Error("completeness must be between 0 and 100");
|
|
2056
|
+
}
|
|
2057
|
+
if (metrics.duration < 0) {
|
|
2058
|
+
throw new Error("duration must be non-negative");
|
|
2059
|
+
}
|
|
2060
|
+
const stmt = this.db.prepare(`
|
|
2061
|
+
INSERT INTO provider_executions (
|
|
2062
|
+
provider, prompt_id, version_id, variant,
|
|
2063
|
+
success, completeness, duration, tokens_used, code_delivered,
|
|
2064
|
+
error_count, warning_count, tests_passed, tests_failed,
|
|
2065
|
+
task_complexity, codebase_size, timestamp
|
|
2066
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
2067
|
+
`);
|
|
2068
|
+
stmt.run(
|
|
2069
|
+
metrics.provider,
|
|
2070
|
+
metrics.promptId,
|
|
2071
|
+
metrics.versionId,
|
|
2072
|
+
metrics.variant || null,
|
|
2073
|
+
metrics.success ? 1 : 0,
|
|
2074
|
+
metrics.completeness,
|
|
2075
|
+
metrics.duration,
|
|
2076
|
+
metrics.tokensUsed || null,
|
|
2077
|
+
metrics.codeDelivered ? 1 : 0,
|
|
2078
|
+
metrics.errorCount || 0,
|
|
2079
|
+
metrics.warningCount || 0,
|
|
2080
|
+
metrics.testsPassed || null,
|
|
2081
|
+
metrics.testsFailed || null,
|
|
2082
|
+
metrics.taskComplexity || null,
|
|
2083
|
+
metrics.codebaseSize || null,
|
|
2084
|
+
metrics.timestamp
|
|
2085
|
+
);
|
|
2086
|
+
await this.invalidateCache(
|
|
2087
|
+
metrics.provider,
|
|
2088
|
+
metrics.promptId,
|
|
2089
|
+
metrics.versionId,
|
|
2090
|
+
metrics.variant
|
|
2091
|
+
);
|
|
2092
|
+
} catch (err) {
|
|
2093
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2094
|
+
throw new Error(`Failed to track provider execution: ${error}`);
|
|
2095
|
+
}
|
|
2096
|
+
}
|
|
2097
|
+
/**
|
|
2098
|
+
* Get provider performance metrics for a specific prompt
|
|
2099
|
+
*
|
|
2100
|
+
* Retrieves comprehensive performance statistics with Bayesian confidence intervals,
|
|
2101
|
+
* trend analysis, and statistical measures.
|
|
2102
|
+
*
|
|
2103
|
+
* @param provider AI provider to analyze
|
|
2104
|
+
* @param promptId Prompt identifier
|
|
2105
|
+
* @param versionId Optional version filter
|
|
2106
|
+
* @param variant Optional variant filter
|
|
2107
|
+
* @returns Performance metrics or null if insufficient data
|
|
2108
|
+
*
|
|
2109
|
+
* @example
|
|
2110
|
+
* ```typescript
|
|
2111
|
+
* const performance = await tracker.getProviderPerformance(
|
|
2112
|
+
* 'claude',
|
|
2113
|
+
* 'typescript-expert',
|
|
2114
|
+
* 'v1.2.0'
|
|
2115
|
+
* );
|
|
2116
|
+
* console.log(`Success rate: ${performance.successRate * 100}%`);
|
|
2117
|
+
* console.log(`95% CI: [${performance.successRateCI[0]}, ${performance.successRateCI[1]}]`);
|
|
2118
|
+
* ```
|
|
2119
|
+
*/
|
|
2120
|
+
async getProviderPerformance(provider, promptId, versionId, variant) {
|
|
2121
|
+
try {
|
|
2122
|
+
if (!this.isValidProvider(provider)) {
|
|
2123
|
+
throw new Error(`Invalid provider: ${provider}`);
|
|
2124
|
+
}
|
|
2125
|
+
const cached = this.getCachedPerformance(provider, promptId, versionId, variant);
|
|
2126
|
+
if (cached && this.isCacheFresh(cached.calculatedAt)) {
|
|
2127
|
+
return cached;
|
|
2128
|
+
}
|
|
2129
|
+
let query = `
|
|
2130
|
+
SELECT * FROM provider_executions
|
|
2131
|
+
WHERE provider = ? AND prompt_id = ?
|
|
2132
|
+
`;
|
|
2133
|
+
const params = [provider, promptId];
|
|
2134
|
+
if (versionId) {
|
|
2135
|
+
query += " AND version_id = ?";
|
|
2136
|
+
params.push(versionId);
|
|
2137
|
+
}
|
|
2138
|
+
if (variant) {
|
|
2139
|
+
query += " AND variant = ?";
|
|
2140
|
+
params.push(variant);
|
|
2141
|
+
}
|
|
2142
|
+
query += " ORDER BY timestamp DESC";
|
|
2143
|
+
const stmt = this.db.prepare(query);
|
|
2144
|
+
const executions = stmt.all(...params);
|
|
2145
|
+
if (executions.length < this.options.minSampleSize) {
|
|
2146
|
+
return null;
|
|
2147
|
+
}
|
|
2148
|
+
const performance = this.calculatePerformanceMetrics(
|
|
2149
|
+
executions,
|
|
2150
|
+
provider,
|
|
2151
|
+
promptId,
|
|
2152
|
+
versionId,
|
|
2153
|
+
variant
|
|
2154
|
+
);
|
|
2155
|
+
this.cachePerformance(performance);
|
|
2156
|
+
return performance;
|
|
2157
|
+
} catch (err) {
|
|
2158
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2159
|
+
throw new Error(`Failed to get provider performance: ${error}`);
|
|
2160
|
+
}
|
|
2161
|
+
}
|
|
2162
|
+
/**
|
|
2163
|
+
* Get comprehensive statistics for a provider
|
|
2164
|
+
*
|
|
2165
|
+
* Aggregates performance across all prompts to provide overall provider characteristics.
|
|
2166
|
+
*
|
|
2167
|
+
* @param provider AI provider to analyze
|
|
2168
|
+
* @returns Provider statistics
|
|
2169
|
+
*
|
|
2170
|
+
* @example
|
|
2171
|
+
* ```typescript
|
|
2172
|
+
* const stats = await tracker.getProviderStats('claude');
|
|
2173
|
+
* console.log(`Reliability: ${stats.reliability * 100}%`);
|
|
2174
|
+
* console.log(`Specializations: ${stats.specialization.join(', ')}`);
|
|
2175
|
+
* ```
|
|
2176
|
+
*/
|
|
2177
|
+
async getProviderStats(provider) {
|
|
2178
|
+
try {
|
|
2179
|
+
if (!this.isValidProvider(provider)) {
|
|
2180
|
+
throw new Error(`Invalid provider: ${provider}`);
|
|
2181
|
+
}
|
|
2182
|
+
const stmt = this.db.prepare(`
|
|
2183
|
+
SELECT
|
|
2184
|
+
COUNT(DISTINCT prompt_id) as total_prompts,
|
|
2185
|
+
COUNT(*) as total_executions,
|
|
2186
|
+
AVG(success) as avg_success_rate,
|
|
2187
|
+
AVG(completeness) as avg_completeness,
|
|
2188
|
+
AVG(duration) as avg_duration,
|
|
2189
|
+
AVG(tokens_used) as avg_tokens
|
|
2190
|
+
FROM provider_executions
|
|
2191
|
+
WHERE provider = ?
|
|
2192
|
+
`);
|
|
2193
|
+
const aggregates = stmt.get(provider);
|
|
2194
|
+
if (aggregates.total_executions === 0) {
|
|
2195
|
+
return this.emptyProviderStats(provider);
|
|
2196
|
+
}
|
|
2197
|
+
const reliability = this.calculateReliabilityScore(provider);
|
|
2198
|
+
const speed = this.calculateSpeedScore(provider);
|
|
2199
|
+
const quality = this.calculateQualityScore(provider);
|
|
2200
|
+
const { bestPromptId, worstPromptId } = this.findBestWorstPrompts(provider);
|
|
2201
|
+
const specialization = this.identifySpecializations(provider);
|
|
2202
|
+
return {
|
|
2203
|
+
provider,
|
|
2204
|
+
totalPrompts: aggregates.total_prompts,
|
|
2205
|
+
totalExecutions: aggregates.total_executions,
|
|
2206
|
+
avgSuccessRate: aggregates.avg_success_rate,
|
|
2207
|
+
avgCompleteness: aggregates.avg_completeness,
|
|
2208
|
+
avgDuration: aggregates.avg_duration,
|
|
2209
|
+
avgTokenEfficiency: aggregates.avg_tokens ? 1 / aggregates.avg_tokens : void 0,
|
|
2210
|
+
reliability,
|
|
2211
|
+
speed,
|
|
2212
|
+
quality,
|
|
2213
|
+
bestPromptId,
|
|
2214
|
+
worstPromptId,
|
|
2215
|
+
specialization,
|
|
2216
|
+
calculatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
2217
|
+
};
|
|
2218
|
+
} catch (err) {
|
|
2219
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2220
|
+
throw new Error(`Failed to get provider stats: ${error}`);
|
|
2221
|
+
}
|
|
2222
|
+
}
|
|
2223
|
+
/**
|
|
2224
|
+
* Get best prompts for a provider
|
|
2225
|
+
*
|
|
2226
|
+
* Uses multi-armed bandit algorithm (Upper Confidence Bound) to balance
|
|
2227
|
+
* exploration vs exploitation when recommending prompts.
|
|
2228
|
+
*
|
|
2229
|
+
* @param provider AI provider
|
|
2230
|
+
* @param limit Maximum number of prompts to return
|
|
2231
|
+
* @returns Array of prompt IDs ranked by UCB score
|
|
2232
|
+
*
|
|
2233
|
+
* @example
|
|
2234
|
+
* ```typescript
|
|
2235
|
+
* const bestPrompts = await tracker.getBestPromptsForProvider('claude', 5);
|
|
2236
|
+
* console.log('Top 5 prompts:', bestPrompts);
|
|
2237
|
+
* ```
|
|
2238
|
+
*/
|
|
2239
|
+
async getBestPromptsForProvider(provider, limit = 10) {
|
|
2240
|
+
try {
|
|
2241
|
+
if (!this.isValidProvider(provider)) {
|
|
2242
|
+
throw new Error(`Invalid provider: ${provider}`);
|
|
2243
|
+
}
|
|
2244
|
+
const stmt = this.db.prepare(`
|
|
2245
|
+
SELECT
|
|
2246
|
+
prompt_id,
|
|
2247
|
+
COUNT(*) as executions,
|
|
2248
|
+
AVG(success) as success_rate,
|
|
2249
|
+
AVG(completeness) as avg_completeness
|
|
2250
|
+
FROM provider_executions
|
|
2251
|
+
WHERE provider = ?
|
|
2252
|
+
GROUP BY prompt_id
|
|
2253
|
+
HAVING executions >= ?
|
|
2254
|
+
`);
|
|
2255
|
+
const prompts = stmt.all(provider, this.options.minSampleSize);
|
|
2256
|
+
if (prompts.length === 0) {
|
|
2257
|
+
return [];
|
|
2258
|
+
}
|
|
2259
|
+
const totalExecutions = prompts.reduce((sum, p) => sum + p.executions, 0);
|
|
2260
|
+
const scored = prompts.map((p) => {
|
|
2261
|
+
const exploitationScore = p.success_rate * 0.5 + p.avg_completeness / 100 * 0.5;
|
|
2262
|
+
const explorationBonus = Math.sqrt(2 * Math.log(totalExecutions) / p.executions) * this.options.explorationRate;
|
|
2263
|
+
const ucbScore = exploitationScore + explorationBonus;
|
|
2264
|
+
return {
|
|
2265
|
+
promptId: p.prompt_id,
|
|
2266
|
+
ucbScore,
|
|
2267
|
+
executions: p.executions
|
|
2268
|
+
};
|
|
2269
|
+
});
|
|
2270
|
+
scored.sort((a, b) => b.ucbScore - a.ucbScore);
|
|
2271
|
+
return scored.slice(0, limit).map((s) => s.promptId);
|
|
2272
|
+
} catch (err) {
|
|
2273
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2274
|
+
throw new Error(`Failed to get best prompts: ${error}`);
|
|
2275
|
+
}
|
|
2276
|
+
}
|
|
2277
|
+
/**
|
|
2278
|
+
* Get provider recommendations for a prompt
|
|
2279
|
+
*
|
|
2280
|
+
* Uses Bayesian inference to recommend the best provider(s) with confidence scores.
|
|
2281
|
+
* Considers success rate, completeness, and duration with uncertainty quantification.
|
|
2282
|
+
*
|
|
2283
|
+
* @param promptId Prompt identifier
|
|
2284
|
+
* @param versionId Optional version filter
|
|
2285
|
+
* @returns Ranked provider recommendations
|
|
2286
|
+
*
|
|
2287
|
+
* @example
|
|
2288
|
+
* ```typescript
|
|
2289
|
+
* const recommendations = await tracker.getProviderRecommendations('typescript-expert');
|
|
2290
|
+
* const best = recommendations[0];
|
|
2291
|
+
* console.log(`Recommend ${best.provider} with ${best.confidence * 100}% confidence`);
|
|
2292
|
+
* ```
|
|
2293
|
+
*/
|
|
2294
|
+
async getProviderRecommendations(promptId, versionId) {
|
|
2295
|
+
try {
|
|
2296
|
+
const recommendations = [];
|
|
2297
|
+
for (const provider of this.validProviders) {
|
|
2298
|
+
const performance = await this.getProviderPerformance(provider, promptId, versionId);
|
|
2299
|
+
if (!performance || performance.sampleSize < this.options.minSampleSize) {
|
|
2300
|
+
continue;
|
|
2301
|
+
}
|
|
2302
|
+
const confidence = this.calculateBayesianConfidence(
|
|
2303
|
+
performance.successfulExecutions,
|
|
2304
|
+
performance.totalExecutions
|
|
2305
|
+
);
|
|
2306
|
+
const reasoning = this.generateRecommendationReasoning(performance);
|
|
2307
|
+
recommendations.push({
|
|
2308
|
+
provider,
|
|
2309
|
+
confidence,
|
|
2310
|
+
expectedSuccess: performance.successRate,
|
|
2311
|
+
expectedCompleteness: performance.avgCompleteness,
|
|
2312
|
+
expectedDuration: performance.avgDuration,
|
|
2313
|
+
reasoning,
|
|
2314
|
+
rank: 0
|
|
2315
|
+
// Will be set after sorting
|
|
2316
|
+
});
|
|
2317
|
+
}
|
|
2318
|
+
recommendations.sort((a, b) => {
|
|
2319
|
+
const scoreA = a.expectedSuccess * (a.expectedCompleteness / 100) * a.confidence;
|
|
2320
|
+
const scoreB = b.expectedSuccess * (b.expectedCompleteness / 100) * b.confidence;
|
|
2321
|
+
return scoreB - scoreA;
|
|
2322
|
+
});
|
|
2323
|
+
recommendations.forEach((rec, index) => {
|
|
2324
|
+
rec.rank = index + 1;
|
|
2325
|
+
});
|
|
2326
|
+
return recommendations;
|
|
2327
|
+
} catch (err) {
|
|
2328
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2329
|
+
throw new Error(`Failed to get provider recommendations: ${error}`);
|
|
2330
|
+
}
|
|
2331
|
+
}
|
|
2332
|
+
/**
|
|
2333
|
+
* Get optimization insights for a provider
|
|
2334
|
+
*
|
|
2335
|
+
* Analyzes performance patterns to identify strengths, weaknesses, opportunities,
|
|
2336
|
+
* and threats (SWOT analysis) for a specific provider and prompt combination.
|
|
2337
|
+
*
|
|
2338
|
+
* @param provider AI provider
|
|
2339
|
+
* @param promptId Prompt identifier
|
|
2340
|
+
* @returns Array of actionable insights
|
|
2341
|
+
*
|
|
2342
|
+
* @example
|
|
2343
|
+
* ```typescript
|
|
2344
|
+
* const insights = await tracker.getProviderOptimizationInsights('claude', 'typescript-expert');
|
|
2345
|
+
* insights.forEach(insight => {
|
|
2346
|
+
* console.log(`[${insight.priority}] ${insight.type}: ${insight.description}`);
|
|
2347
|
+
* });
|
|
2348
|
+
* ```
|
|
2349
|
+
*/
|
|
2350
|
+
async getProviderOptimizationInsights(provider, promptId) {
|
|
2351
|
+
try {
|
|
2352
|
+
const insights = [];
|
|
2353
|
+
const performance = await this.getProviderPerformance(provider, promptId);
|
|
2354
|
+
if (!performance) {
|
|
2355
|
+
return insights;
|
|
2356
|
+
}
|
|
2357
|
+
if (performance.successRate >= 0.9) {
|
|
2358
|
+
insights.push({
|
|
2359
|
+
provider,
|
|
2360
|
+
promptId,
|
|
2361
|
+
type: "strength",
|
|
2362
|
+
priority: "low",
|
|
2363
|
+
description: "Excellent success rate",
|
|
2364
|
+
impact: `${(performance.successRate * 100).toFixed(1)}% success rate indicates high reliability`,
|
|
2365
|
+
recommendation: "Consider using this provider as primary for this prompt",
|
|
2366
|
+
confidence: 0.9
|
|
2367
|
+
});
|
|
2368
|
+
} else if (performance.successRate < 0.7) {
|
|
2369
|
+
insights.push({
|
|
2370
|
+
provider,
|
|
2371
|
+
promptId,
|
|
2372
|
+
type: "weakness",
|
|
2373
|
+
priority: "high",
|
|
2374
|
+
description: "Low success rate",
|
|
2375
|
+
impact: `${(performance.successRate * 100).toFixed(1)}% success rate indicates reliability issues`,
|
|
2376
|
+
recommendation: "Review prompt structure, consider alternative provider, or add error handling",
|
|
2377
|
+
confidence: 0.85
|
|
2378
|
+
});
|
|
2379
|
+
}
|
|
2380
|
+
if (performance.stdDevCompleteness > 20) {
|
|
2381
|
+
insights.push({
|
|
2382
|
+
provider,
|
|
2383
|
+
promptId,
|
|
2384
|
+
type: "weakness",
|
|
2385
|
+
priority: "medium",
|
|
2386
|
+
description: "Inconsistent output quality",
|
|
2387
|
+
impact: `High variance (\u03C3=${performance.stdDevCompleteness.toFixed(1)}) in completeness scores`,
|
|
2388
|
+
recommendation: "Add more specific instructions or examples to improve consistency",
|
|
2389
|
+
confidence: 0.75
|
|
2390
|
+
});
|
|
2391
|
+
} else if (performance.stdDevCompleteness < 10) {
|
|
2392
|
+
insights.push({
|
|
2393
|
+
provider,
|
|
2394
|
+
promptId,
|
|
2395
|
+
type: "strength",
|
|
2396
|
+
priority: "low",
|
|
2397
|
+
description: "Highly consistent performance",
|
|
2398
|
+
impact: `Low variance (\u03C3=${performance.stdDevCompleteness.toFixed(1)}) indicates predictable outputs`,
|
|
2399
|
+
recommendation: "This provider-prompt combination is production-ready",
|
|
2400
|
+
confidence: 0.85
|
|
2401
|
+
});
|
|
2402
|
+
}
|
|
2403
|
+
if (performance.performanceTrend === "declining") {
|
|
2404
|
+
insights.push({
|
|
2405
|
+
provider,
|
|
2406
|
+
promptId,
|
|
2407
|
+
type: "threat",
|
|
2408
|
+
priority: performance.trendStrength > 0.7 ? "critical" : "medium",
|
|
2409
|
+
description: "Performance degradation detected",
|
|
2410
|
+
impact: "Quality or reliability is declining over time",
|
|
2411
|
+
recommendation: "Investigate recent changes, consider prompt refresh or provider update",
|
|
2412
|
+
confidence: performance.trendStrength
|
|
2413
|
+
});
|
|
2414
|
+
} else if (performance.performanceTrend === "improving") {
|
|
2415
|
+
insights.push({
|
|
2416
|
+
provider,
|
|
2417
|
+
promptId,
|
|
2418
|
+
type: "opportunity",
|
|
2419
|
+
priority: "low",
|
|
2420
|
+
description: "Performance improving over time",
|
|
2421
|
+
impact: "Quality or reliability is increasing",
|
|
2422
|
+
recommendation: "Continue current approach, consider expanding usage",
|
|
2423
|
+
confidence: performance.trendStrength
|
|
2424
|
+
});
|
|
2425
|
+
}
|
|
2426
|
+
if (performance.avgDuration > 180) {
|
|
2427
|
+
insights.push({
|
|
2428
|
+
provider,
|
|
2429
|
+
promptId,
|
|
2430
|
+
type: "weakness",
|
|
2431
|
+
priority: "medium",
|
|
2432
|
+
description: "Slow execution time",
|
|
2433
|
+
impact: `Average duration of ${performance.avgDuration.toFixed(0)}s may impact user experience`,
|
|
2434
|
+
recommendation: "Consider optimizing prompt length, task scope, or using faster provider",
|
|
2435
|
+
confidence: 0.8
|
|
2436
|
+
});
|
|
2437
|
+
} else if (performance.avgDuration < 60) {
|
|
2438
|
+
insights.push({
|
|
2439
|
+
provider,
|
|
2440
|
+
promptId,
|
|
2441
|
+
type: "strength",
|
|
2442
|
+
priority: "low",
|
|
2443
|
+
description: "Fast execution time",
|
|
2444
|
+
impact: `Average duration of ${performance.avgDuration.toFixed(0)}s enables quick iterations`,
|
|
2445
|
+
recommendation: "Consider this provider for time-sensitive tasks",
|
|
2446
|
+
confidence: 0.85
|
|
2447
|
+
});
|
|
2448
|
+
}
|
|
2449
|
+
const priorityOrder = { critical: 0, high: 1, medium: 2, low: 3 };
|
|
2450
|
+
insights.sort((a, b) => priorityOrder[a.priority] - priorityOrder[b.priority]);
|
|
2451
|
+
return insights;
|
|
2452
|
+
} catch (err) {
|
|
2453
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2454
|
+
throw new Error(`Failed to get optimization insights: ${error}`);
|
|
2455
|
+
}
|
|
2456
|
+
}
|
|
2457
|
+
/**
|
|
2458
|
+
* Get provider-specific insights for a prompt
|
|
2459
|
+
*
|
|
2460
|
+
* Returns insights about which providers and strategies work best for this prompt.
|
|
2461
|
+
* This method is used by the auto-optimizer to make informed decisions.
|
|
2462
|
+
*
|
|
2463
|
+
* @param promptId Prompt identifier
|
|
2464
|
+
* @returns Array of provider insights with performance data
|
|
2465
|
+
*/
|
|
2466
|
+
async getProviderInsights(promptId) {
|
|
2467
|
+
try {
|
|
2468
|
+
const insights = [];
|
|
2469
|
+
for (const provider of this.validProviders) {
|
|
2470
|
+
const performance = await this.getProviderPerformance(provider, promptId);
|
|
2471
|
+
if (!performance || performance.sampleSize < this.options.minSampleSize) {
|
|
2472
|
+
continue;
|
|
2473
|
+
}
|
|
2474
|
+
const performanceScore = performance.successRate * 50 + // 50% weight for success rate
|
|
2475
|
+
performance.avgCompleteness / 100 * 30 + // 30% weight for completeness
|
|
2476
|
+
Math.max(0, 100 - performance.avgDuration / 3) * 20;
|
|
2477
|
+
let bestStrategy = "conservative";
|
|
2478
|
+
if (performance.successRate > 0.9 && performance.avgCompleteness > 90) {
|
|
2479
|
+
bestStrategy = "innovative";
|
|
2480
|
+
} else if (performance.successRate > 0.8 && performance.stdDevCompleteness < 15) {
|
|
2481
|
+
bestStrategy = "balanced";
|
|
2482
|
+
}
|
|
2483
|
+
insights.push({
|
|
2484
|
+
provider,
|
|
2485
|
+
bestStrategy,
|
|
2486
|
+
performanceScore: Math.round(performanceScore * 100) / 100,
|
|
2487
|
+
sampleSize: performance.sampleSize
|
|
2488
|
+
});
|
|
2489
|
+
}
|
|
2490
|
+
insights.sort((a, b) => b.performanceScore - a.performanceScore);
|
|
2491
|
+
return insights;
|
|
2492
|
+
} catch (err) {
|
|
2493
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2494
|
+
throw new Error(`Failed to get provider insights: ${error}`);
|
|
2495
|
+
}
|
|
2496
|
+
}
|
|
2497
|
+
/**
|
|
2498
|
+
* Compare providers for a specific prompt
|
|
2499
|
+
*
|
|
2500
|
+
* Performs comprehensive comparison across all providers with statistical rigor.
|
|
2501
|
+
*
|
|
2502
|
+
* @param promptId Prompt identifier
|
|
2503
|
+
* @param versionId Optional version filter
|
|
2504
|
+
* @returns Comparison result with recommendations
|
|
2505
|
+
*/
|
|
2506
|
+
async compareProviders(promptId, versionId) {
|
|
2507
|
+
try {
|
|
2508
|
+
const recommendations = await this.getProviderRecommendations(promptId, versionId);
|
|
2509
|
+
if (recommendations.length === 0) {
|
|
2510
|
+
throw new Error("Insufficient data for provider comparison");
|
|
2511
|
+
}
|
|
2512
|
+
const best = recommendations[0];
|
|
2513
|
+
if (!best) {
|
|
2514
|
+
throw new Error("No best provider found in recommendations");
|
|
2515
|
+
}
|
|
2516
|
+
const recommendationText = this.generateComparisonRecommendation(recommendations);
|
|
2517
|
+
const confidence = recommendations.reduce((sum, rec) => sum + rec.confidence, 0) / recommendations.length;
|
|
2518
|
+
return {
|
|
2519
|
+
promptId,
|
|
2520
|
+
versionId: versionId || "all",
|
|
2521
|
+
providers: recommendations,
|
|
2522
|
+
bestProvider: best.provider,
|
|
2523
|
+
recommendation: recommendationText,
|
|
2524
|
+
confidence
|
|
2525
|
+
};
|
|
2526
|
+
} catch (err) {
|
|
2527
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
2528
|
+
throw new Error(`Failed to compare providers: ${error}`);
|
|
2529
|
+
}
|
|
2530
|
+
}
|
|
2531
|
+
/**
|
|
2532
|
+
* Close database connection
|
|
2533
|
+
*
|
|
2534
|
+
* Should be called when done with the tracker to properly release resources.
|
|
2535
|
+
*/
|
|
2536
|
+
close() {
|
|
2537
|
+
this.db.close();
|
|
2538
|
+
}
|
|
2539
|
+
// ============================================================================
|
|
2540
|
+
// PRIVATE HELPER METHODS
|
|
2541
|
+
// ============================================================================
|
|
2542
|
+
/**
|
|
2543
|
+
* Whether a provider id should be accepted for tracking. Registry-known and
|
|
2544
|
+
* user-defined providers are valid [B7]. If the registry could not be loaded
|
|
2545
|
+
* (empty set), any non-empty id is accepted so tracking degrades open rather
|
|
2546
|
+
* than rejecting every provider.
|
|
2547
|
+
*/
|
|
2548
|
+
isValidProvider(provider) {
|
|
2549
|
+
if (this.validProviderSet.size === 0) {
|
|
2550
|
+
return typeof provider === "string" && provider.length > 0;
|
|
2551
|
+
}
|
|
2552
|
+
return this.validProviderSet.has(provider);
|
|
2553
|
+
}
|
|
2554
|
+
/**
|
|
2555
|
+
* Calculate comprehensive performance metrics from raw executions
|
|
2556
|
+
*/
|
|
2557
|
+
calculatePerformanceMetrics(executions, provider, promptId, versionId, variant) {
|
|
2558
|
+
const totalExecutions = executions.length;
|
|
2559
|
+
const successfulExecutions = executions.filter((e) => e.success === 1).length;
|
|
2560
|
+
const successRate = successfulExecutions / totalExecutions;
|
|
2561
|
+
const completenessValues = executions.map((e) => e.completeness).sort((a, b) => a - b);
|
|
2562
|
+
const avgCompleteness = completenessValues.reduce((sum, v) => sum + v, 0) / completenessValues.length;
|
|
2563
|
+
const minCompleteness = completenessValues[0] ?? 0;
|
|
2564
|
+
const maxCompleteness = completenessValues[completenessValues.length - 1] ?? 0;
|
|
2565
|
+
const medianCompleteness = this.calculateMedian(completenessValues);
|
|
2566
|
+
const p95Completeness = this.calculatePercentile(completenessValues, 0.95);
|
|
2567
|
+
const stdDevCompleteness = this.calculateStdDev(completenessValues, avgCompleteness);
|
|
2568
|
+
const durationValues = executions.map((e) => e.duration);
|
|
2569
|
+
const avgDuration = durationValues.reduce((sum, v) => sum + v, 0) / durationValues.length;
|
|
2570
|
+
const codeDeliveredCount = executions.filter((e) => e.code_delivered === 1).length;
|
|
2571
|
+
const avgCodeDeliveryRate = codeDeliveredCount / totalExecutions;
|
|
2572
|
+
const successRateCI = this.calculateBayesianCI(successfulExecutions, totalExecutions);
|
|
2573
|
+
const completenessCI = this.calculateNormalCI(
|
|
2574
|
+
avgCompleteness,
|
|
2575
|
+
stdDevCompleteness,
|
|
2576
|
+
totalExecutions
|
|
2577
|
+
);
|
|
2578
|
+
const { trend, strength } = this.analyzeTrend(executions);
|
|
2579
|
+
const timestamps = executions.map((e) => e.timestamp).sort();
|
|
2580
|
+
const firstSeenAt = timestamps[0] ?? (/* @__PURE__ */ new Date()).toISOString();
|
|
2581
|
+
const lastSeenAt = timestamps[timestamps.length - 1] ?? (/* @__PURE__ */ new Date()).toISOString();
|
|
2582
|
+
return {
|
|
2583
|
+
provider,
|
|
2584
|
+
promptId,
|
|
2585
|
+
versionId: versionId || "all",
|
|
2586
|
+
variant,
|
|
2587
|
+
totalExecutions,
|
|
2588
|
+
successfulExecutions,
|
|
2589
|
+
successRate,
|
|
2590
|
+
avgCompleteness,
|
|
2591
|
+
avgDuration,
|
|
2592
|
+
avgCodeDeliveryRate,
|
|
2593
|
+
stdDevCompleteness,
|
|
2594
|
+
medianCompleteness,
|
|
2595
|
+
p95Completeness,
|
|
2596
|
+
minCompleteness,
|
|
2597
|
+
maxCompleteness,
|
|
2598
|
+
successRateCI,
|
|
2599
|
+
completenessCI,
|
|
2600
|
+
performanceTrend: trend,
|
|
2601
|
+
trendStrength: strength,
|
|
2602
|
+
sampleSize: totalExecutions,
|
|
2603
|
+
firstSeenAt,
|
|
2604
|
+
lastSeenAt,
|
|
2605
|
+
calculatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
2606
|
+
};
|
|
2607
|
+
}
|
|
2608
|
+
/**
|
|
2609
|
+
* Calculate Bayesian confidence interval for success rate
|
|
2610
|
+
* Uses Beta distribution (conjugate prior for Binomial)
|
|
2611
|
+
*/
|
|
2612
|
+
calculateBayesianCI(successes, total) {
|
|
2613
|
+
const alpha = successes + 0.5;
|
|
2614
|
+
const beta = total - successes + 0.5;
|
|
2615
|
+
const mean = alpha / (alpha + beta);
|
|
2616
|
+
const variance = alpha * beta / ((alpha + beta) ** 2 * (alpha + beta + 1));
|
|
2617
|
+
const stdDev = Math.sqrt(variance);
|
|
2618
|
+
const z = 1.96;
|
|
2619
|
+
const lower = Math.max(0, mean - z * stdDev);
|
|
2620
|
+
const upper = Math.min(1, mean + z * stdDev);
|
|
2621
|
+
return [lower, upper];
|
|
2622
|
+
}
|
|
2623
|
+
/**
|
|
2624
|
+
* Calculate normal confidence interval for continuous metrics
|
|
2625
|
+
*/
|
|
2626
|
+
calculateNormalCI(mean, stdDev, n) {
|
|
2627
|
+
const z = 1.96;
|
|
2628
|
+
const margin = z * (stdDev / Math.sqrt(n));
|
|
2629
|
+
return [Math.max(0, mean - margin), Math.min(100, mean + margin)];
|
|
2630
|
+
}
|
|
2631
|
+
/**
|
|
2632
|
+
* Calculate Bayesian confidence score
|
|
2633
|
+
*/
|
|
2634
|
+
calculateBayesianConfidence(successes, total) {
|
|
2635
|
+
const alpha = successes + 0.5;
|
|
2636
|
+
const beta = total - successes + 0.5;
|
|
2637
|
+
const mean = alpha / (alpha + beta);
|
|
2638
|
+
const sampleConfidence = Math.min(1, total / 20);
|
|
2639
|
+
return mean * sampleConfidence;
|
|
2640
|
+
}
|
|
2641
|
+
/**
|
|
2642
|
+
* Analyze performance trend using linear regression
|
|
2643
|
+
*/
|
|
2644
|
+
analyzeTrend(executions) {
|
|
2645
|
+
if (executions.length < 5) {
|
|
2646
|
+
return { trend: "insufficient_data", strength: 0 };
|
|
2647
|
+
}
|
|
2648
|
+
const sorted = [...executions].sort(
|
|
2649
|
+
(a, b) => new Date(a.timestamp).getTime() - new Date(b.timestamp).getTime()
|
|
2650
|
+
);
|
|
2651
|
+
const x = sorted.map((_, i) => i);
|
|
2652
|
+
const y = sorted.map((e) => e.completeness);
|
|
2653
|
+
const { slope, rSquared } = this.linearRegression(x, y);
|
|
2654
|
+
let trend;
|
|
2655
|
+
if (Math.abs(slope) < 1) {
|
|
2656
|
+
trend = "stable";
|
|
2657
|
+
} else if (slope > 0) {
|
|
2658
|
+
trend = "improving";
|
|
2659
|
+
} else {
|
|
2660
|
+
trend = "declining";
|
|
2661
|
+
}
|
|
2662
|
+
return { trend, strength: rSquared };
|
|
2663
|
+
}
|
|
2664
|
+
/**
|
|
2665
|
+
* Perform linear regression
|
|
2666
|
+
*/
|
|
2667
|
+
linearRegression(x, y) {
|
|
2668
|
+
const n = x.length;
|
|
2669
|
+
const sumX = x.reduce((sum, v) => sum + v, 0);
|
|
2670
|
+
const sumY = y.reduce((sum, v) => sum + v, 0);
|
|
2671
|
+
const sumXY = x.reduce((sum, v, i) => sum + v * (y[i] ?? 0), 0);
|
|
2672
|
+
const sumXX = x.reduce((sum, v) => sum + v * v, 0);
|
|
2673
|
+
const sumYY = y.reduce((sum, v) => sum + v * v, 0);
|
|
2674
|
+
const slope = (n * sumXY - sumX * sumY) / (n * sumXX - sumX * sumX);
|
|
2675
|
+
const intercept = (sumY - slope * sumX) / n;
|
|
2676
|
+
const meanY = sumY / n;
|
|
2677
|
+
const ssTotal = sumYY - n * meanY * meanY;
|
|
2678
|
+
const ssResidual = y.reduce((sum, yi, i) => {
|
|
2679
|
+
const predicted = intercept + slope * (x[i] ?? 0);
|
|
2680
|
+
return sum + (yi - predicted) ** 2;
|
|
2681
|
+
}, 0);
|
|
2682
|
+
const rSquared = 1 - ssResidual / ssTotal;
|
|
2683
|
+
return { slope, intercept, rSquared };
|
|
2684
|
+
}
|
|
2685
|
+
/**
|
|
2686
|
+
* Calculate median
|
|
2687
|
+
*/
|
|
2688
|
+
calculateMedian(sorted) {
|
|
2689
|
+
const mid = Math.floor(sorted.length / 2);
|
|
2690
|
+
const lower = sorted[mid - 1] ?? 0;
|
|
2691
|
+
const upper = sorted[mid] ?? 0;
|
|
2692
|
+
const single = sorted[mid] ?? 0;
|
|
2693
|
+
return sorted.length % 2 === 0 ? (lower + upper) / 2 : single;
|
|
2694
|
+
}
|
|
2695
|
+
/**
|
|
2696
|
+
* Calculate percentile
|
|
2697
|
+
*/
|
|
2698
|
+
calculatePercentile(sorted, p) {
|
|
2699
|
+
const index = Math.ceil(sorted.length * p) - 1;
|
|
2700
|
+
return sorted[Math.max(0, Math.min(index, sorted.length - 1))] ?? 0;
|
|
2701
|
+
}
|
|
2702
|
+
/**
|
|
2703
|
+
* Calculate standard deviation
|
|
2704
|
+
*/
|
|
2705
|
+
calculateStdDev(values, mean) {
|
|
2706
|
+
const variance = values.reduce((sum, v) => sum + (v - mean) ** 2, 0) / values.length;
|
|
2707
|
+
return Math.sqrt(variance);
|
|
2708
|
+
}
|
|
2709
|
+
/**
|
|
2710
|
+
* Calculate reliability score (consistency)
|
|
2711
|
+
*/
|
|
2712
|
+
calculateReliabilityScore(provider) {
|
|
2713
|
+
const stmt = this.db.prepare(`
|
|
2714
|
+
SELECT AVG(completeness) as mean,
|
|
2715
|
+
(AVG(completeness * completeness) - AVG(completeness) * AVG(completeness)) as variance
|
|
2716
|
+
FROM provider_executions
|
|
2717
|
+
WHERE provider = ?
|
|
2718
|
+
`);
|
|
2719
|
+
const result = stmt.get(provider);
|
|
2720
|
+
if (!result || result.mean === null || result.variance === null) return 0;
|
|
2721
|
+
const stdDev = Math.sqrt(result.variance);
|
|
2722
|
+
const coefficientOfVariation = stdDev / result.mean;
|
|
2723
|
+
return Math.max(0, 1 - coefficientOfVariation);
|
|
2724
|
+
}
|
|
2725
|
+
/**
|
|
2726
|
+
* Calculate speed score (normalized)
|
|
2727
|
+
*/
|
|
2728
|
+
calculateSpeedScore(provider) {
|
|
2729
|
+
const stmt = this.db.prepare(`
|
|
2730
|
+
SELECT AVG(duration) as avg_duration
|
|
2731
|
+
FROM provider_executions
|
|
2732
|
+
WHERE provider = ?
|
|
2733
|
+
`);
|
|
2734
|
+
const result = stmt.get(provider);
|
|
2735
|
+
if (!result || result.avg_duration === null) return 0;
|
|
2736
|
+
const normalized = 1 - Math.min(1, (result.avg_duration - 60) / 240);
|
|
2737
|
+
return Math.max(0, normalized);
|
|
2738
|
+
}
|
|
2739
|
+
/**
|
|
2740
|
+
* Calculate quality score
|
|
2741
|
+
*/
|
|
2742
|
+
calculateQualityScore(provider) {
|
|
2743
|
+
const stmt = this.db.prepare(`
|
|
2744
|
+
SELECT AVG(success) as success_rate,
|
|
2745
|
+
AVG(completeness) as avg_completeness
|
|
2746
|
+
FROM provider_executions
|
|
2747
|
+
WHERE provider = ?
|
|
2748
|
+
`);
|
|
2749
|
+
const result = stmt.get(provider);
|
|
2750
|
+
if (!result) return 0;
|
|
2751
|
+
const functionality = (typeof result.avg_completeness === "number" ? result.avg_completeness : 0) * 0.5 + (typeof result.success_rate === "number" ? result.success_rate * 100 : 0) * 0.5;
|
|
2752
|
+
const base = {
|
|
2753
|
+
functionality_correctness: functionality,
|
|
2754
|
+
scalability_adaptability: result.avg_completeness,
|
|
2755
|
+
code_quality_maintainability: result.avg_completeness,
|
|
2756
|
+
security_robustness: Math.min(85, result.avg_completeness)
|
|
2757
|
+
};
|
|
2758
|
+
const scores = buildCriterionScores(base);
|
|
2759
|
+
const agg = aggregateScores(scores);
|
|
2760
|
+
return agg.score / 100;
|
|
2761
|
+
}
|
|
2762
|
+
/**
|
|
2763
|
+
* Find best and worst prompts for a provider
|
|
2764
|
+
*/
|
|
2765
|
+
findBestWorstPrompts(provider) {
|
|
2766
|
+
const stmt = this.db.prepare(`
|
|
2767
|
+
SELECT prompt_id,
|
|
2768
|
+
AVG(success) as success_rate,
|
|
2769
|
+
AVG(completeness) as avg_completeness
|
|
2770
|
+
FROM provider_executions
|
|
2771
|
+
WHERE provider = ?
|
|
2772
|
+
GROUP BY prompt_id
|
|
2773
|
+
HAVING COUNT(*) >= ?
|
|
2774
|
+
`);
|
|
2775
|
+
const prompts = stmt.all(provider, this.options.minSampleSize);
|
|
2776
|
+
if (prompts.length === 0) {
|
|
2777
|
+
return { bestPromptId: null, worstPromptId: null };
|
|
2778
|
+
}
|
|
2779
|
+
const scored = prompts.map((p) => ({
|
|
2780
|
+
promptId: p.prompt_id,
|
|
2781
|
+
score: (p.success_rate ?? 0) * 0.5 + (p.avg_completeness ?? 0) / 100 * 0.5
|
|
2782
|
+
}));
|
|
2783
|
+
scored.sort((a, b) => b.score - a.score);
|
|
2784
|
+
return {
|
|
2785
|
+
bestPromptId: scored[0]?.promptId ?? null,
|
|
2786
|
+
worstPromptId: scored[scored.length - 1]?.promptId ?? null
|
|
2787
|
+
};
|
|
2788
|
+
}
|
|
2789
|
+
/**
|
|
2790
|
+
* Identify provider specializations
|
|
2791
|
+
*/
|
|
2792
|
+
identifySpecializations(provider) {
|
|
2793
|
+
const specializations = [];
|
|
2794
|
+
const complexStmt = this.db.prepare(`
|
|
2795
|
+
SELECT task_complexity, AVG(success) as success_rate
|
|
2796
|
+
FROM provider_executions
|
|
2797
|
+
WHERE provider = ? AND task_complexity IS NOT NULL
|
|
2798
|
+
GROUP BY task_complexity
|
|
2799
|
+
`);
|
|
2800
|
+
const complexResults = complexStmt.all(provider);
|
|
2801
|
+
for (const result of complexResults) {
|
|
2802
|
+
if (result.success_rate > 0.85) {
|
|
2803
|
+
specializations.push(`${result.task_complexity} tasks`);
|
|
2804
|
+
}
|
|
2805
|
+
}
|
|
2806
|
+
if (specializations.length === 0) {
|
|
2807
|
+
specializations.push("general purpose");
|
|
2808
|
+
}
|
|
2809
|
+
return specializations;
|
|
2810
|
+
}
|
|
2811
|
+
/**
|
|
2812
|
+
* Generate recommendation reasoning
|
|
2813
|
+
*/
|
|
2814
|
+
generateRecommendationReasoning(performance) {
|
|
2815
|
+
const reasoning = [];
|
|
2816
|
+
if (performance.successRate >= 0.9) {
|
|
2817
|
+
reasoning.push(`High success rate (${(performance.successRate * 100).toFixed(1)}%)`);
|
|
2818
|
+
}
|
|
2819
|
+
if (performance.avgCompleteness >= 85) {
|
|
2820
|
+
reasoning.push(`Excellent completeness (${performance.avgCompleteness.toFixed(1)}/100)`);
|
|
2821
|
+
}
|
|
2822
|
+
if (performance.stdDevCompleteness < 10) {
|
|
2823
|
+
reasoning.push("Highly consistent performance");
|
|
2824
|
+
}
|
|
2825
|
+
if (performance.avgDuration < 60) {
|
|
2826
|
+
reasoning.push(`Fast execution (${performance.avgDuration.toFixed(0)}s)`);
|
|
2827
|
+
}
|
|
2828
|
+
if (performance.performanceTrend === "improving") {
|
|
2829
|
+
reasoning.push("Performance improving over time");
|
|
2830
|
+
} else if (performance.performanceTrend === "declining") {
|
|
2831
|
+
reasoning.push(" Performance declining - needs attention");
|
|
2832
|
+
}
|
|
2833
|
+
if (reasoning.length === 0) {
|
|
2834
|
+
reasoning.push("Adequate performance");
|
|
2835
|
+
}
|
|
2836
|
+
return reasoning;
|
|
2837
|
+
}
|
|
2838
|
+
/**
|
|
2839
|
+
* Generate comparison recommendation text
|
|
2840
|
+
*/
|
|
2841
|
+
generateComparisonRecommendation(recommendations) {
|
|
2842
|
+
if (recommendations.length === 0) {
|
|
2843
|
+
return "No data available for comparison";
|
|
2844
|
+
}
|
|
2845
|
+
const best = recommendations[0];
|
|
2846
|
+
if (!best) {
|
|
2847
|
+
return "No best provider available";
|
|
2848
|
+
}
|
|
2849
|
+
const parts = [];
|
|
2850
|
+
parts.push(`Recommend ${best.provider} as primary provider`);
|
|
2851
|
+
parts.push(`Expected success: ${(best.expectedSuccess * 100).toFixed(1)}%`);
|
|
2852
|
+
parts.push(`Expected completeness: ${best.expectedCompleteness.toFixed(1)}/100`);
|
|
2853
|
+
parts.push(`Confidence: ${(best.confidence * 100).toFixed(1)}%`);
|
|
2854
|
+
if (recommendations.length > 1) {
|
|
2855
|
+
const second = recommendations[1];
|
|
2856
|
+
if (second) {
|
|
2857
|
+
const gap = (best.expectedSuccess - second.expectedSuccess) * 100;
|
|
2858
|
+
if (gap < 5) {
|
|
2859
|
+
parts.push(
|
|
2860
|
+
`Note: ${second.provider} is a close alternative (${gap.toFixed(1)}% difference)`
|
|
2861
|
+
);
|
|
2862
|
+
}
|
|
2863
|
+
}
|
|
2864
|
+
}
|
|
2865
|
+
return parts.join(". ");
|
|
2866
|
+
}
|
|
2867
|
+
/**
|
|
2868
|
+
* Empty provider stats for when no data exists
|
|
2869
|
+
*/
|
|
2870
|
+
emptyProviderStats(provider) {
|
|
2871
|
+
return {
|
|
2872
|
+
provider,
|
|
2873
|
+
totalPrompts: 0,
|
|
2874
|
+
totalExecutions: 0,
|
|
2875
|
+
avgSuccessRate: 0,
|
|
2876
|
+
avgCompleteness: 0,
|
|
2877
|
+
avgDuration: 0,
|
|
2878
|
+
reliability: 0,
|
|
2879
|
+
speed: 0,
|
|
2880
|
+
quality: 0,
|
|
2881
|
+
bestPromptId: null,
|
|
2882
|
+
worstPromptId: null,
|
|
2883
|
+
specialization: [],
|
|
2884
|
+
calculatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
2885
|
+
};
|
|
2886
|
+
}
|
|
2887
|
+
/**
|
|
2888
|
+
* Cache performance results
|
|
2889
|
+
*/
|
|
2890
|
+
cachePerformance(performance) {
|
|
2891
|
+
const stmt = this.db.prepare(`
|
|
2892
|
+
INSERT OR REPLACE INTO provider_stats_cache
|
|
2893
|
+
(provider, prompt_id, version_id, variant, total_executions, success_rate,
|
|
2894
|
+
avg_completeness, avg_duration, std_dev_completeness, median_completeness,
|
|
2895
|
+
performance_trend, trend_strength, calculated_at, sample_size)
|
|
2896
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
2897
|
+
`);
|
|
2898
|
+
stmt.run(
|
|
2899
|
+
performance.provider,
|
|
2900
|
+
performance.promptId,
|
|
2901
|
+
performance.versionId || "",
|
|
2902
|
+
performance.variant || "",
|
|
2903
|
+
performance.totalExecutions,
|
|
2904
|
+
performance.successRate,
|
|
2905
|
+
performance.avgCompleteness,
|
|
2906
|
+
performance.avgDuration,
|
|
2907
|
+
performance.stdDevCompleteness,
|
|
2908
|
+
performance.medianCompleteness,
|
|
2909
|
+
performance.performanceTrend,
|
|
2910
|
+
performance.trendStrength,
|
|
2911
|
+
performance.calculatedAt,
|
|
2912
|
+
performance.sampleSize
|
|
2913
|
+
);
|
|
2914
|
+
}
|
|
2915
|
+
/**
|
|
2916
|
+
* Get cached performance if available and fresh
|
|
2917
|
+
*/
|
|
2918
|
+
getCachedPerformance(provider, promptId, versionId, variant) {
|
|
2919
|
+
const stmt = this.db.prepare(`
|
|
2920
|
+
SELECT * FROM provider_stats_cache
|
|
2921
|
+
WHERE provider = ? AND prompt_id = ? AND version_id = ? AND variant = ?
|
|
2922
|
+
`);
|
|
2923
|
+
const row = stmt.get(provider, promptId, versionId || "", variant || "");
|
|
2924
|
+
if (!row) return null;
|
|
2925
|
+
return {
|
|
2926
|
+
provider: row.provider,
|
|
2927
|
+
promptId: row.prompt_id,
|
|
2928
|
+
versionId: row.version_id || "",
|
|
2929
|
+
variant: row.variant || void 0,
|
|
2930
|
+
totalExecutions: row.total_executions,
|
|
2931
|
+
successfulExecutions: Math.round(row.success_rate * row.total_executions),
|
|
2932
|
+
successRate: row.success_rate,
|
|
2933
|
+
avgCompleteness: row.avg_completeness,
|
|
2934
|
+
avgDuration: row.avg_duration,
|
|
2935
|
+
avgCodeDeliveryRate: 0,
|
|
2936
|
+
// Not cached
|
|
2937
|
+
stdDevCompleteness: row.std_dev_completeness,
|
|
2938
|
+
medianCompleteness: row.median_completeness,
|
|
2939
|
+
p95Completeness: 0,
|
|
2940
|
+
// Not cached
|
|
2941
|
+
minCompleteness: 0,
|
|
2942
|
+
// Not cached
|
|
2943
|
+
maxCompleteness: 100,
|
|
2944
|
+
// Not cached
|
|
2945
|
+
successRateCI: [0, 1],
|
|
2946
|
+
// Not cached
|
|
2947
|
+
completenessCI: [0, 100],
|
|
2948
|
+
// Not cached
|
|
2949
|
+
performanceTrend: row.performance_trend,
|
|
2950
|
+
trendStrength: row.trend_strength,
|
|
2951
|
+
sampleSize: row.sample_size,
|
|
2952
|
+
firstSeenAt: "",
|
|
2953
|
+
// Not cached
|
|
2954
|
+
lastSeenAt: "",
|
|
2955
|
+
// Not cached
|
|
2956
|
+
calculatedAt: row.calculated_at
|
|
2957
|
+
};
|
|
2958
|
+
}
|
|
2959
|
+
/**
|
|
2960
|
+
* Check if cached result is fresh (within 1 hour)
|
|
2961
|
+
*/
|
|
2962
|
+
isCacheFresh(calculatedAt) {
|
|
2963
|
+
const cached = new Date(calculatedAt).getTime();
|
|
2964
|
+
const now = Date.now();
|
|
2965
|
+
const ageMs = now - cached;
|
|
2966
|
+
const maxAgeMs = 60 * 60 * 1e3;
|
|
2967
|
+
return ageMs < maxAgeMs;
|
|
2968
|
+
}
|
|
2969
|
+
/**
|
|
2970
|
+
* Invalidate cache for a specific combination
|
|
2971
|
+
*/
|
|
2972
|
+
async invalidateCache(provider, promptId, versionId, variant) {
|
|
2973
|
+
const stmt = this.db.prepare(`
|
|
2974
|
+
DELETE FROM provider_stats_cache
|
|
2975
|
+
WHERE provider = ? AND prompt_id = ? AND version_id = ? AND variant = ?
|
|
2976
|
+
`);
|
|
2977
|
+
stmt.run(provider, promptId, versionId || "", variant || "");
|
|
2978
|
+
}
|
|
2979
|
+
};
|
|
2980
|
+
|
|
2981
|
+
// src/prompts/auto-optimizer.ts
|
|
2982
|
+
var AutomaticPromptOptimizer = class {
|
|
2983
|
+
store;
|
|
2984
|
+
manager;
|
|
2985
|
+
evaluator;
|
|
2986
|
+
optimizer;
|
|
2987
|
+
variantGenerator;
|
|
2988
|
+
providerTracker;
|
|
2989
|
+
autoTuner;
|
|
2990
|
+
activationAuthority;
|
|
2991
|
+
telemetry;
|
|
2992
|
+
lineage;
|
|
2993
|
+
options;
|
|
2994
|
+
constructor(options = {}) {
|
|
2995
|
+
this.telemetry = options.telemetryLogger || TelemetryLogger.getInstance(options);
|
|
2996
|
+
this.store = options.promptStore || new PromptStore(options);
|
|
2997
|
+
this.manager = options.promptManager || new PromptManager(options);
|
|
2998
|
+
this.evaluator = new PromptEvaluator({
|
|
2999
|
+
poeticDir: options.poeticDir,
|
|
3000
|
+
telemetryLogger: this.telemetry
|
|
3001
|
+
});
|
|
3002
|
+
this.optimizer = new PromptOptimizer({
|
|
3003
|
+
poeticDir: options.poeticDir,
|
|
3004
|
+
telemetryLogger: this.telemetry
|
|
3005
|
+
});
|
|
3006
|
+
this.variantGenerator = new VariantGenerator();
|
|
3007
|
+
this.providerTracker = new ProviderTracker(options);
|
|
3008
|
+
this.autoTuner = new AutoTuningEngine(options);
|
|
3009
|
+
this.activationAuthority = options.activationAuthority ?? new TunerActivationAuthority(this.autoTuner, this.telemetry);
|
|
3010
|
+
this.lineage = options.lineageTracker || new LineageTracker(options);
|
|
3011
|
+
this.options = {
|
|
3012
|
+
poeticDir: resolvePoeticDirFromContext(options.poeticDir),
|
|
3013
|
+
telemetryLogger: this.telemetry,
|
|
3014
|
+
minSampleSize: options.minSampleSize ?? 10,
|
|
3015
|
+
improvementThreshold: options.improvementThreshold ?? 5,
|
|
3016
|
+
maxVariantsPerOptimization: options.maxVariantsPerOptimization ?? 3,
|
|
3017
|
+
autoActivateBestVariant: options.autoActivateBestVariant ?? false,
|
|
3018
|
+
enableAutoTuning: options.enableAutoTuning ?? false,
|
|
3019
|
+
requireManualReview: options.requireManualReview ?? true,
|
|
3020
|
+
rollbackOnRegression: options.rollbackOnRegression ?? true
|
|
3021
|
+
};
|
|
3022
|
+
}
|
|
3023
|
+
/**
|
|
3024
|
+
* Analyze a prompt and generate optimization recommendations
|
|
3025
|
+
*/
|
|
3026
|
+
async analyzePrompt(promptId) {
|
|
3027
|
+
try {
|
|
3028
|
+
this.manager.getOrThrow(promptId);
|
|
3029
|
+
const activeVersion = this.manager.getActiveVersionOrThrow(promptId);
|
|
3030
|
+
const currentPerformance = await this.evaluator.calculatePerformanceMetrics(activeVersion);
|
|
3031
|
+
if (currentPerformance.sampleSize < this.options.minSampleSize) {
|
|
3032
|
+
return {
|
|
3033
|
+
promptId,
|
|
3034
|
+
currentVersionId: activeVersion.id,
|
|
3035
|
+
currentPerformance,
|
|
3036
|
+
suggestions: [],
|
|
3037
|
+
proposedVariants: [],
|
|
3038
|
+
providerInsights: [],
|
|
3039
|
+
action: "needs_more_data",
|
|
3040
|
+
rationale: `Insufficient data (${currentPerformance.sampleSize}/${this.options.minSampleSize} samples). Continue running tasks to gather more performance data.`,
|
|
3041
|
+
estimatedImpact: {
|
|
3042
|
+
successRateImprovement: 0,
|
|
3043
|
+
completenessImprovement: 0,
|
|
3044
|
+
durationImprovement: 0
|
|
3045
|
+
},
|
|
3046
|
+
confidence: 0,
|
|
3047
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
3048
|
+
};
|
|
3049
|
+
}
|
|
3050
|
+
const analysis = await this.optimizer.analyze(activeVersion.id);
|
|
3051
|
+
if (analysis.overallScore >= 85 && analysis.suggestions.length === 0) {
|
|
3052
|
+
return {
|
|
3053
|
+
promptId,
|
|
3054
|
+
currentVersionId: activeVersion.id,
|
|
3055
|
+
currentPerformance,
|
|
3056
|
+
suggestions: [],
|
|
3057
|
+
proposedVariants: [],
|
|
3058
|
+
providerInsights: await this.getProviderInsights(promptId),
|
|
3059
|
+
action: "no_action",
|
|
3060
|
+
rationale: "Prompt is performing well. No optimization needed at this time.",
|
|
3061
|
+
estimatedImpact: {
|
|
3062
|
+
successRateImprovement: 0,
|
|
3063
|
+
completenessImprovement: 0,
|
|
3064
|
+
durationImprovement: 0
|
|
3065
|
+
},
|
|
3066
|
+
confidence: 0.9,
|
|
3067
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
3068
|
+
};
|
|
3069
|
+
}
|
|
3070
|
+
const bestVersion = await this.optimizer.recommendBestVersion(promptId);
|
|
3071
|
+
if (bestVersion.recommendedVersion !== activeVersion.id) {
|
|
3072
|
+
const recommendedMetrics = await this.evaluator.calculatePerformanceMetrics(
|
|
3073
|
+
this.store.getVersion(promptId, bestVersion.recommendedVersion)
|
|
3074
|
+
);
|
|
3075
|
+
const improvement = (recommendedMetrics.successRate - currentPerformance.successRate) * 100;
|
|
3076
|
+
if (improvement >= this.options.improvementThreshold) {
|
|
3077
|
+
return {
|
|
3078
|
+
promptId,
|
|
3079
|
+
currentVersionId: activeVersion.id,
|
|
3080
|
+
currentPerformance,
|
|
3081
|
+
suggestions: analysis.suggestions,
|
|
3082
|
+
proposedVariants: [],
|
|
3083
|
+
providerInsights: await this.getProviderInsights(promptId),
|
|
3084
|
+
action: "activate_existing",
|
|
3085
|
+
rationale: `Existing version ${bestVersion.recommendedVersion} performs ${improvement.toFixed(1)}% better. Recommend activating it.`,
|
|
3086
|
+
estimatedImpact: {
|
|
3087
|
+
successRateImprovement: improvement,
|
|
3088
|
+
completenessImprovement: recommendedMetrics.avgCompleteness - currentPerformance.avgCompleteness,
|
|
3089
|
+
durationImprovement: currentPerformance.avgDuration - recommendedMetrics.avgDuration
|
|
3090
|
+
},
|
|
3091
|
+
confidence: 0.8,
|
|
3092
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
3093
|
+
};
|
|
3094
|
+
}
|
|
3095
|
+
}
|
|
3096
|
+
const generatedVariants = await this.variantGenerator.generateVariants(
|
|
3097
|
+
activeVersion,
|
|
3098
|
+
this.options.maxVariantsPerOptimization,
|
|
3099
|
+
{
|
|
3100
|
+
strategies: this.selectStrategiesForSuggestions(analysis.suggestions),
|
|
3101
|
+
targetMetric: this.determineTargetMetric(analysis.suggestions)
|
|
3102
|
+
}
|
|
3103
|
+
);
|
|
3104
|
+
const proposedVariants = generatedVariants.map(
|
|
3105
|
+
(variant, index) => this.toProposedVariant(activeVersion.id, index, variant)
|
|
3106
|
+
);
|
|
3107
|
+
const estimatedImpact = this.estimateImpact(analysis.suggestions);
|
|
3108
|
+
const confidence = this.calculateConfidence(
|
|
3109
|
+
currentPerformance.sampleSize,
|
|
3110
|
+
analysis.suggestions
|
|
3111
|
+
);
|
|
3112
|
+
return {
|
|
3113
|
+
promptId,
|
|
3114
|
+
currentVersionId: activeVersion.id,
|
|
3115
|
+
currentPerformance,
|
|
3116
|
+
suggestions: analysis.suggestions,
|
|
3117
|
+
proposedVariants,
|
|
3118
|
+
providerInsights: await this.getProviderInsights(promptId),
|
|
3119
|
+
action: "generate_variants",
|
|
3120
|
+
rationale: `Generated ${proposedVariants.length} suggested prompt revision proposals based on performance analysis. Estimated ${estimatedImpact.successRateImprovement.toFixed(1)}% improvement.`,
|
|
3121
|
+
estimatedImpact,
|
|
3122
|
+
confidence,
|
|
3123
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
3124
|
+
};
|
|
3125
|
+
} catch (error) {
|
|
3126
|
+
throw new Error(`Failed to analyze prompt ${promptId}: ${error.message}`);
|
|
3127
|
+
}
|
|
3128
|
+
}
|
|
3129
|
+
/**
|
|
3130
|
+
* Execute automatic optimization for a prompt
|
|
3131
|
+
*/
|
|
3132
|
+
async optimizePrompt(promptId, recommendation) {
|
|
3133
|
+
try {
|
|
3134
|
+
if (!recommendation) {
|
|
3135
|
+
recommendation = await this.analyzePrompt(promptId);
|
|
3136
|
+
}
|
|
3137
|
+
const result = {
|
|
3138
|
+
promptId,
|
|
3139
|
+
success: false,
|
|
3140
|
+
variantsGenerated: 0,
|
|
3141
|
+
variantsActivated: 0,
|
|
3142
|
+
improvements: [],
|
|
3143
|
+
warnings: [],
|
|
3144
|
+
errors: [],
|
|
3145
|
+
suggestedPromptRevisions: [],
|
|
3146
|
+
beforeMetrics: recommendation.currentPerformance,
|
|
3147
|
+
executedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
3148
|
+
};
|
|
3149
|
+
switch (recommendation.action) {
|
|
3150
|
+
case "needs_more_data":
|
|
3151
|
+
result.warnings.push(recommendation.rationale);
|
|
3152
|
+
result.success = true;
|
|
3153
|
+
break;
|
|
3154
|
+
case "no_action":
|
|
3155
|
+
result.improvements.push(recommendation.rationale);
|
|
3156
|
+
result.success = true;
|
|
3157
|
+
break;
|
|
3158
|
+
case "activate_existing":
|
|
3159
|
+
await this.activateExistingVariant(promptId, recommendation, result);
|
|
3160
|
+
break;
|
|
3161
|
+
case "generate_variants":
|
|
3162
|
+
await this.generateAndTestVariants(promptId, recommendation, result);
|
|
3163
|
+
break;
|
|
3164
|
+
}
|
|
3165
|
+
return result;
|
|
3166
|
+
} catch (error) {
|
|
3167
|
+
throw new Error(`Failed to optimize prompt ${promptId}: ${error.message}`);
|
|
3168
|
+
}
|
|
3169
|
+
}
|
|
3170
|
+
/**
|
|
3171
|
+
* Batch optimize all prompts that need optimization
|
|
3172
|
+
*/
|
|
3173
|
+
async optimizeAllPrompts() {
|
|
3174
|
+
try {
|
|
3175
|
+
const prompts = this.manager.list({ status: "active" });
|
|
3176
|
+
const results = [];
|
|
3177
|
+
const errors = [];
|
|
3178
|
+
let optimized = 0;
|
|
3179
|
+
for (const prompt of prompts) {
|
|
3180
|
+
try {
|
|
3181
|
+
const recommendation = await this.analyzePrompt(prompt.id);
|
|
3182
|
+
if (recommendation.action === "generate_variants" || recommendation.action === "activate_existing") {
|
|
3183
|
+
const result = await this.optimizePrompt(prompt.id, recommendation);
|
|
3184
|
+
results.push(result);
|
|
3185
|
+
if (result.success) {
|
|
3186
|
+
optimized++;
|
|
3187
|
+
}
|
|
3188
|
+
}
|
|
3189
|
+
} catch (error) {
|
|
3190
|
+
errors.push({
|
|
3191
|
+
promptId: prompt.id,
|
|
3192
|
+
_error: error.message
|
|
3193
|
+
});
|
|
3194
|
+
}
|
|
3195
|
+
}
|
|
3196
|
+
return {
|
|
3197
|
+
analyzed: prompts.length,
|
|
3198
|
+
optimized,
|
|
3199
|
+
results,
|
|
3200
|
+
errors
|
|
3201
|
+
};
|
|
3202
|
+
} catch (error) {
|
|
3203
|
+
throw new Error(`Failed to optimize all prompts: ${error.message}`);
|
|
3204
|
+
}
|
|
3205
|
+
}
|
|
3206
|
+
/**
|
|
3207
|
+
* Monitor prompt performance and trigger optimization when needed
|
|
3208
|
+
*/
|
|
3209
|
+
async monitorAndOptimize(promptId) {
|
|
3210
|
+
try {
|
|
3211
|
+
this.manager.getOrThrow(promptId);
|
|
3212
|
+
const activeVersion = this.manager.getActiveVersionOrThrow(promptId);
|
|
3213
|
+
const performance = await this.evaluator.calculatePerformanceMetrics(activeVersion);
|
|
3214
|
+
const shouldOptimize = await this.shouldTriggerOptimization(promptId, performance);
|
|
3215
|
+
if (!shouldOptimize.trigger) {
|
|
3216
|
+
return {
|
|
3217
|
+
optimizationTriggered: false,
|
|
3218
|
+
reason: shouldOptimize.reason
|
|
3219
|
+
};
|
|
3220
|
+
}
|
|
3221
|
+
const result = await this.optimizePrompt(promptId);
|
|
3222
|
+
return {
|
|
3223
|
+
optimizationTriggered: true,
|
|
3224
|
+
reason: shouldOptimize.reason,
|
|
3225
|
+
result
|
|
3226
|
+
};
|
|
3227
|
+
} catch (error) {
|
|
3228
|
+
throw new Error(
|
|
3229
|
+
`Failed to monitor and optimize prompt ${promptId}: ${error.message}`
|
|
3230
|
+
);
|
|
3231
|
+
}
|
|
3232
|
+
}
|
|
3233
|
+
/**
|
|
3234
|
+
* Rollback to previous version if regression detected
|
|
3235
|
+
*/
|
|
3236
|
+
async rollbackIfRegression(promptId) {
|
|
3237
|
+
try {
|
|
3238
|
+
if (!this.options.rollbackOnRegression) {
|
|
3239
|
+
return { rollbackPerformed: false };
|
|
3240
|
+
}
|
|
3241
|
+
this.manager.getOrThrow(promptId);
|
|
3242
|
+
const versions = this.manager.listVersions(promptId);
|
|
3243
|
+
if (versions.length < 2) {
|
|
3244
|
+
return { rollbackPerformed: false };
|
|
3245
|
+
}
|
|
3246
|
+
const current = versions.find((v) => v.isActive);
|
|
3247
|
+
const previous = versions.filter((v) => !v.isActive)[0];
|
|
3248
|
+
if (!current || !previous) {
|
|
3249
|
+
return { rollbackPerformed: false };
|
|
3250
|
+
}
|
|
3251
|
+
const currentPerf = await this.evaluator.calculatePerformanceMetrics(current);
|
|
3252
|
+
const previousPerf = await this.evaluator.calculatePerformanceMetrics(previous);
|
|
3253
|
+
const successRateDelta = (currentPerf.successRate - previousPerf.successRate) * 100;
|
|
3254
|
+
const completenessDelta = currentPerf.avgCompleteness - previousPerf.avgCompleteness;
|
|
3255
|
+
const hasRegression = successRateDelta < -10 || completenessDelta < -15;
|
|
3256
|
+
if (hasRegression && currentPerf.sampleSize >= 5) {
|
|
3257
|
+
const reason = `Detected regression: success rate ${successRateDelta.toFixed(1)}%, completeness ${completenessDelta.toFixed(1)}`;
|
|
3258
|
+
await this.activateThroughAuthority({
|
|
3259
|
+
promptId,
|
|
3260
|
+
sourceVersionId: current.id,
|
|
3261
|
+
candidateVersionId: previous.id,
|
|
3262
|
+
improvementPercentage: 0,
|
|
3263
|
+
confidence: 0.9,
|
|
3264
|
+
reason,
|
|
3265
|
+
skipSafetyChecks: true
|
|
3266
|
+
});
|
|
3267
|
+
return {
|
|
3268
|
+
rollbackPerformed: true,
|
|
3269
|
+
previousVersionId: previous.id,
|
|
3270
|
+
reason
|
|
3271
|
+
};
|
|
3272
|
+
}
|
|
3273
|
+
return { rollbackPerformed: false };
|
|
3274
|
+
} catch (error) {
|
|
3275
|
+
throw new Error(
|
|
3276
|
+
`Failed to check for regression on prompt ${promptId}: ${error.message}`
|
|
3277
|
+
);
|
|
3278
|
+
}
|
|
3279
|
+
}
|
|
3280
|
+
/**
|
|
3281
|
+
* Close all connections
|
|
3282
|
+
*/
|
|
3283
|
+
close() {
|
|
3284
|
+
this.store.close();
|
|
3285
|
+
this.manager.close();
|
|
3286
|
+
this.evaluator.close();
|
|
3287
|
+
this.optimizer.close();
|
|
3288
|
+
this.variantGenerator.close();
|
|
3289
|
+
this.providerTracker.close();
|
|
3290
|
+
this.autoTuner.close();
|
|
3291
|
+
this.telemetry.close();
|
|
3292
|
+
}
|
|
3293
|
+
// Private helper methods
|
|
3294
|
+
/**
|
|
3295
|
+
* Activate an existing better-performing variant
|
|
3296
|
+
*/
|
|
3297
|
+
async activateExistingVariant(promptId, _recommendation, result) {
|
|
3298
|
+
try {
|
|
3299
|
+
const bestVersion = await this.optimizer.recommendBestVersion(promptId);
|
|
3300
|
+
if (this.options.requireManualReview && !this.options.enableAutoTuning) {
|
|
3301
|
+
result.warnings.push(
|
|
3302
|
+
`Manual review required before activating version ${bestVersion.recommendedVersion}`
|
|
3303
|
+
);
|
|
3304
|
+
result.success = true;
|
|
3305
|
+
return;
|
|
3306
|
+
}
|
|
3307
|
+
const activeVersion = this.manager.getActiveVersionOrThrow(promptId);
|
|
3308
|
+
await this.activateThroughAuthority({
|
|
3309
|
+
promptId,
|
|
3310
|
+
sourceVersionId: activeVersion.id,
|
|
3311
|
+
candidateVersionId: bestVersion.recommendedVersion,
|
|
3312
|
+
improvementPercentage: _recommendation.estimatedImpact.successRateImprovement,
|
|
3313
|
+
confidence: _recommendation.confidence,
|
|
3314
|
+
reason: bestVersion.rationale
|
|
3315
|
+
});
|
|
3316
|
+
result.variantsActivated = 1;
|
|
3317
|
+
result.improvements.push(
|
|
3318
|
+
`Activated version ${bestVersion.recommendedVersion}: ${bestVersion.rationale}`
|
|
3319
|
+
);
|
|
3320
|
+
const newVersion = this.manager.getVersion(promptId, bestVersion.recommendedVersion);
|
|
3321
|
+
if (newVersion) {
|
|
3322
|
+
result.afterMetrics = await this.evaluator.calculatePerformanceMetrics(newVersion);
|
|
3323
|
+
result.performanceChange = (result.afterMetrics.successRate - result.beforeMetrics.successRate) * 100;
|
|
3324
|
+
}
|
|
3325
|
+
result.success = true;
|
|
3326
|
+
} catch (error) {
|
|
3327
|
+
result.errors.push(error.message);
|
|
3328
|
+
result.success = false;
|
|
3329
|
+
}
|
|
3330
|
+
}
|
|
3331
|
+
/**
|
|
3332
|
+
* Generate and test new variants
|
|
3333
|
+
*/
|
|
3334
|
+
async generateAndTestVariants(promptId, recommendation, result) {
|
|
3335
|
+
try {
|
|
3336
|
+
const activeVersion = this.manager.getActiveVersionOrThrow(promptId);
|
|
3337
|
+
const cycleId = this.createOptimizationCycleId(promptId);
|
|
3338
|
+
const generatedVersions = [];
|
|
3339
|
+
for (const proposed of recommendation.proposedVariants) {
|
|
3340
|
+
try {
|
|
3341
|
+
const generationPrompt = this.buildGenerationPrompt(proposed);
|
|
3342
|
+
const newVersion = await this.manager.createVersion({
|
|
3343
|
+
promptId,
|
|
3344
|
+
content: proposed.content,
|
|
3345
|
+
changelog: `Suggested prompt revision (${proposed.variantId}): ${proposed.description}`,
|
|
3346
|
+
author: "auto-optimizer",
|
|
3347
|
+
activate: false,
|
|
3348
|
+
// Don't activate yet (unless auto-tuning is enabled)
|
|
3349
|
+
isGenerated: true,
|
|
3350
|
+
generationModel: "poetic-auto-optimizer",
|
|
3351
|
+
generationPrompt
|
|
3352
|
+
});
|
|
3353
|
+
generatedVersions.push(newVersion.id);
|
|
3354
|
+
try {
|
|
3355
|
+
await this.lineage.trackMutation(
|
|
3356
|
+
activeVersion.id,
|
|
3357
|
+
newVersion.id,
|
|
3358
|
+
this.mapMutations(proposed.mutations, proposed.confidence),
|
|
3359
|
+
cycleId
|
|
3360
|
+
);
|
|
3361
|
+
} catch (lineageError) {
|
|
3362
|
+
emitWarning(
|
|
3363
|
+
"WS-003",
|
|
3364
|
+
{
|
|
3365
|
+
taskId: "auto-optimizer",
|
|
3366
|
+
subsystem: "prompts",
|
|
3367
|
+
executionPhase: "execution"
|
|
3368
|
+
},
|
|
3369
|
+
{
|
|
3370
|
+
severity: "WARN" /* WARN */,
|
|
3371
|
+
metadata: {
|
|
3372
|
+
error: lineageError.message,
|
|
3373
|
+
operation: "trackMutation",
|
|
3374
|
+
versionId: newVersion.id,
|
|
3375
|
+
promptId
|
|
3376
|
+
}
|
|
3377
|
+
}
|
|
3378
|
+
);
|
|
3379
|
+
result.warnings.push(
|
|
3380
|
+
`Lineage tracking failed for ${newVersion.id}: ${lineageError.message}`
|
|
3381
|
+
);
|
|
3382
|
+
}
|
|
3383
|
+
result.variantsGenerated++;
|
|
3384
|
+
const revisionSummary = `Suggested prompt revision ${newVersion.id}: ${proposed.description} (expected ${proposed.expectedImprovement})`;
|
|
3385
|
+
result.suggestedPromptRevisions.push(revisionSummary);
|
|
3386
|
+
result.improvements.push(revisionSummary);
|
|
3387
|
+
} catch (error) {
|
|
3388
|
+
result.errors.push(
|
|
3389
|
+
`Failed to generate suggested prompt revision: ${error.message}`
|
|
3390
|
+
);
|
|
3391
|
+
}
|
|
3392
|
+
}
|
|
3393
|
+
if (this.options.enableAutoTuning && this.options.autoActivateBestVariant && recommendation.confidence >= 0.8 && result.variantsGenerated > 0 && generatedVersions.length > 0) {
|
|
3394
|
+
const variantToActivate = generatedVersions[0];
|
|
3395
|
+
if (!variantToActivate) {
|
|
3396
|
+
result.errors.push("No variant generated for activation");
|
|
3397
|
+
} else {
|
|
3398
|
+
try {
|
|
3399
|
+
await this.activateThroughAuthority({
|
|
3400
|
+
promptId,
|
|
3401
|
+
sourceVersionId: activeVersion.id,
|
|
3402
|
+
candidateVersionId: variantToActivate,
|
|
3403
|
+
improvementPercentage: recommendation.estimatedImpact.successRateImprovement,
|
|
3404
|
+
confidence: recommendation.confidence,
|
|
3405
|
+
reason: recommendation.rationale
|
|
3406
|
+
});
|
|
3407
|
+
result.variantsActivated++;
|
|
3408
|
+
result.improvements.push(
|
|
3409
|
+
`Activated suggested prompt revision ${variantToActivate} (${recommendation.rationale})`
|
|
3410
|
+
);
|
|
3411
|
+
} catch (activationError) {
|
|
3412
|
+
result.warnings.push(
|
|
3413
|
+
`Failed to auto-activate variant ${variantToActivate}: ${activationError.message}`
|
|
3414
|
+
);
|
|
3415
|
+
}
|
|
3416
|
+
}
|
|
3417
|
+
} else if ((this.options.enableAutoTuning || this.options.autoActivateBestVariant) && recommendation.confidence < 0.8) {
|
|
3418
|
+
result.warnings.push(
|
|
3419
|
+
`Confidence too low (${(recommendation.confidence * 100).toFixed(0)}%) for auto-activation. Suggested prompt revisions generated but require manual review.`
|
|
3420
|
+
);
|
|
3421
|
+
} else if (result.variantsGenerated > 0) {
|
|
3422
|
+
result.warnings.push(
|
|
3423
|
+
"Auto-activation disabled - suggested prompt revisions generated but need testing before activation"
|
|
3424
|
+
);
|
|
3425
|
+
}
|
|
3426
|
+
result.success = result.variantsGenerated > 0;
|
|
3427
|
+
} catch (error) {
|
|
3428
|
+
result.errors.push(error.message);
|
|
3429
|
+
result.success = false;
|
|
3430
|
+
}
|
|
3431
|
+
}
|
|
3432
|
+
async activateThroughAuthority(request) {
|
|
3433
|
+
const result = await this.activationAuthority.activate({
|
|
3434
|
+
...request,
|
|
3435
|
+
safetyChecksPassed: null
|
|
3436
|
+
});
|
|
3437
|
+
if (!result.activated) {
|
|
3438
|
+
throw new Error(result.reason || "Activation authority declined activation");
|
|
3439
|
+
}
|
|
3440
|
+
const recordId = result.record?.id;
|
|
3441
|
+
if (recordId && this.activationAuthority.scheduleMonitoring) {
|
|
3442
|
+
this.activationAuthority.scheduleMonitoring(recordId);
|
|
3443
|
+
}
|
|
3444
|
+
}
|
|
3445
|
+
toProposedVariant(baseVersionId, index, variant) {
|
|
3446
|
+
const variantId = `${baseVersionId}-opt-${index + 1}`;
|
|
3447
|
+
const changes = variant.mutations.length ? variant.mutations.map((mutation) => `- [${mutation.strategy}] ${mutation.description}`).join("\n") : "- No structural mutations recorded";
|
|
3448
|
+
return {
|
|
3449
|
+
variantId,
|
|
3450
|
+
description: variant.testHypothesis,
|
|
3451
|
+
changes,
|
|
3452
|
+
expectedImprovement: variant.expectedImprovement,
|
|
3453
|
+
content: variant.content,
|
|
3454
|
+
mutations: variant.mutations,
|
|
3455
|
+
hypothesis: variant.testHypothesis,
|
|
3456
|
+
confidence: variant.confidence
|
|
3457
|
+
};
|
|
3458
|
+
}
|
|
3459
|
+
selectStrategiesForSuggestions(suggestions) {
|
|
3460
|
+
const mapping = {
|
|
3461
|
+
improve_clarity: ["clarity", "structure"],
|
|
3462
|
+
add_context: ["context", "examples"],
|
|
3463
|
+
remove_redundancy: ["conciseness", "structure"],
|
|
3464
|
+
update_strategy: ["hybrid", "clarity"],
|
|
3465
|
+
split_prompt: ["structure", "constraints"],
|
|
3466
|
+
merge_prompts: ["structure", "clarity"]
|
|
3467
|
+
};
|
|
3468
|
+
const selected = /* @__PURE__ */ new Set();
|
|
3469
|
+
for (const suggestion of suggestions) {
|
|
3470
|
+
const strategies = mapping[suggestion.type] || ["hybrid"];
|
|
3471
|
+
strategies.forEach((strategy) => {
|
|
3472
|
+
selected.add(strategy);
|
|
3473
|
+
});
|
|
3474
|
+
}
|
|
3475
|
+
if (selected.size === 0) {
|
|
3476
|
+
["clarity", "context", "structure"].forEach((strategy) => {
|
|
3477
|
+
selected.add(strategy);
|
|
3478
|
+
});
|
|
3479
|
+
}
|
|
3480
|
+
return Array.from(selected).slice(0, this.options.maxVariantsPerOptimization);
|
|
3481
|
+
}
|
|
3482
|
+
determineTargetMetric(suggestions) {
|
|
3483
|
+
if (suggestions.some(
|
|
3484
|
+
(s) => s.affectedMetrics.some((metric) => metric.toLowerCase().includes("success"))
|
|
3485
|
+
)) {
|
|
3486
|
+
return "success_rate";
|
|
3487
|
+
}
|
|
3488
|
+
if (suggestions.some(
|
|
3489
|
+
(s) => s.affectedMetrics.some((metric) => metric.toLowerCase().includes("completeness"))
|
|
3490
|
+
)) {
|
|
3491
|
+
return "completeness";
|
|
3492
|
+
}
|
|
3493
|
+
if (suggestions.some(
|
|
3494
|
+
(s) => s.affectedMetrics.some((metric) => metric.toLowerCase().includes("duration"))
|
|
3495
|
+
)) {
|
|
3496
|
+
return "duration";
|
|
3497
|
+
}
|
|
3498
|
+
return "balanced";
|
|
3499
|
+
}
|
|
3500
|
+
buildGenerationPrompt(proposed) {
|
|
3501
|
+
const header = `Hypothesis: ${proposed.hypothesis}`;
|
|
3502
|
+
const improvement = `Expected impact: ${proposed.expectedImprovement}`;
|
|
3503
|
+
const mutations = proposed.mutations.length ? proposed.mutations.map(
|
|
3504
|
+
(mutation) => `? ${mutation.strategy.toUpperCase()}: ${mutation.description} (${mutation.rationale})`
|
|
3505
|
+
).join("\n") : "? No explicit mutations captured";
|
|
3506
|
+
return `${header}
|
|
3507
|
+
${improvement}
|
|
3508
|
+
Mutations:
|
|
3509
|
+
${mutations}`;
|
|
3510
|
+
}
|
|
3511
|
+
mapMutations(mutations, confidence) {
|
|
3512
|
+
if (mutations.length === 0) {
|
|
3513
|
+
return [
|
|
3514
|
+
{
|
|
3515
|
+
type: "other",
|
|
3516
|
+
description: "No explicit mutation recorded",
|
|
3517
|
+
details: "Auto-optimizer generated a holistic revision without discrete mutations",
|
|
3518
|
+
impact: Math.max(0.2, Math.min(1, confidence))
|
|
3519
|
+
}
|
|
3520
|
+
];
|
|
3521
|
+
}
|
|
3522
|
+
return mutations.map((mutation) => ({
|
|
3523
|
+
type: this.mapMutationType(mutation.strategy),
|
|
3524
|
+
description: mutation.description,
|
|
3525
|
+
details: mutation.rationale || `${mutation.before} ? ${mutation.after}`,
|
|
3526
|
+
impact: Math.max(0.2, Math.min(1, confidence))
|
|
3527
|
+
}));
|
|
3528
|
+
}
|
|
3529
|
+
mapMutationType(strategy) {
|
|
3530
|
+
switch (strategy) {
|
|
3531
|
+
case "clarity":
|
|
3532
|
+
return "clarify_instructions";
|
|
3533
|
+
case "context":
|
|
3534
|
+
case "examples":
|
|
3535
|
+
return "add_context";
|
|
3536
|
+
case "conciseness":
|
|
3537
|
+
return "remove_redundancy";
|
|
3538
|
+
case "structure":
|
|
3539
|
+
case "constraints":
|
|
3540
|
+
return "optimize_structure";
|
|
3541
|
+
case "tone":
|
|
3542
|
+
return "adjust_tone";
|
|
3543
|
+
default:
|
|
3544
|
+
return "other";
|
|
3545
|
+
}
|
|
3546
|
+
}
|
|
3547
|
+
createOptimizationCycleId(promptId) {
|
|
3548
|
+
return `auto-opt-${promptId}-${Date.now()}`;
|
|
3549
|
+
}
|
|
3550
|
+
/**
|
|
3551
|
+
* Get provider-specific insights
|
|
3552
|
+
*/
|
|
3553
|
+
async getProviderInsights(promptId) {
|
|
3554
|
+
try {
|
|
3555
|
+
const insights = await this.providerTracker.getProviderInsights(promptId);
|
|
3556
|
+
return insights.map((i) => ({
|
|
3557
|
+
provider: i.provider,
|
|
3558
|
+
bestStrategy: i.bestStrategy,
|
|
3559
|
+
performanceScore: i.performanceScore,
|
|
3560
|
+
sampleSize: i.sampleSize
|
|
3561
|
+
}));
|
|
3562
|
+
} catch (error) {
|
|
3563
|
+
emitWarning(
|
|
3564
|
+
"WS-003",
|
|
3565
|
+
{
|
|
3566
|
+
taskId: "auto-optimizer",
|
|
3567
|
+
subsystem: "prompts",
|
|
3568
|
+
executionPhase: "execution"
|
|
3569
|
+
},
|
|
3570
|
+
{
|
|
3571
|
+
severity: "WARN" /* WARN */,
|
|
3572
|
+
metadata: {
|
|
3573
|
+
error: error.message,
|
|
3574
|
+
operation: "getProviderInsights",
|
|
3575
|
+
promptId
|
|
3576
|
+
}
|
|
3577
|
+
}
|
|
3578
|
+
);
|
|
3579
|
+
return [];
|
|
3580
|
+
}
|
|
3581
|
+
}
|
|
3582
|
+
/**
|
|
3583
|
+
* Estimate impact of suggested improvements
|
|
3584
|
+
*/
|
|
3585
|
+
estimateImpact(suggestions) {
|
|
3586
|
+
let successRateImprovement = 0;
|
|
3587
|
+
let completenessImprovement = 0;
|
|
3588
|
+
let durationImprovement = 0;
|
|
3589
|
+
for (const suggestion of suggestions) {
|
|
3590
|
+
switch (suggestion.type) {
|
|
3591
|
+
case "improve_clarity":
|
|
3592
|
+
successRateImprovement += 10;
|
|
3593
|
+
completenessImprovement += 8;
|
|
3594
|
+
break;
|
|
3595
|
+
case "add_context":
|
|
3596
|
+
completenessImprovement += 12;
|
|
3597
|
+
successRateImprovement += 5;
|
|
3598
|
+
break;
|
|
3599
|
+
case "remove_redundancy":
|
|
3600
|
+
durationImprovement += 15;
|
|
3601
|
+
break;
|
|
3602
|
+
case "update_strategy":
|
|
3603
|
+
successRateImprovement += 8;
|
|
3604
|
+
completenessImprovement += 5;
|
|
3605
|
+
break;
|
|
3606
|
+
case "split_prompt":
|
|
3607
|
+
durationImprovement += 30;
|
|
3608
|
+
break;
|
|
3609
|
+
}
|
|
3610
|
+
}
|
|
3611
|
+
return {
|
|
3612
|
+
successRateImprovement: Math.min(successRateImprovement, 25),
|
|
3613
|
+
// Cap at 25%
|
|
3614
|
+
completenessImprovement: Math.min(completenessImprovement, 20),
|
|
3615
|
+
// Cap at 20 points
|
|
3616
|
+
durationImprovement: Math.min(durationImprovement, 40)
|
|
3617
|
+
// Cap at 40%
|
|
3618
|
+
};
|
|
3619
|
+
}
|
|
3620
|
+
/**
|
|
3621
|
+
* Calculate confidence in recommendations
|
|
3622
|
+
*/
|
|
3623
|
+
calculateConfidence(sampleSize, suggestions) {
|
|
3624
|
+
let confidence = Math.min(sampleSize / 50, 1);
|
|
3625
|
+
const highPriority = suggestions.filter((s) => s.priority === "high").length;
|
|
3626
|
+
const criticalPriority = suggestions.filter((s) => s.priority === "critical").length;
|
|
3627
|
+
if (criticalPriority > 0) {
|
|
3628
|
+
confidence *= 0.9;
|
|
3629
|
+
} else if (highPriority > 2) {
|
|
3630
|
+
confidence *= 0.85;
|
|
3631
|
+
}
|
|
3632
|
+
return Math.max(0.3, Math.min(1, confidence));
|
|
3633
|
+
}
|
|
3634
|
+
/**
|
|
3635
|
+
* Determine if optimization should be triggered
|
|
3636
|
+
*/
|
|
3637
|
+
async shouldTriggerOptimization(_promptId, performance) {
|
|
3638
|
+
if (performance.sampleSize < this.options.minSampleSize) {
|
|
3639
|
+
return {
|
|
3640
|
+
trigger: false,
|
|
3641
|
+
reason: `Insufficient data (${performance.sampleSize}/${this.options.minSampleSize})`
|
|
3642
|
+
};
|
|
3643
|
+
}
|
|
3644
|
+
if (performance.successRate < 0.7) {
|
|
3645
|
+
return {
|
|
3646
|
+
trigger: true,
|
|
3647
|
+
reason: `Low success rate (${(performance.successRate * 100).toFixed(1)}%)`
|
|
3648
|
+
};
|
|
3649
|
+
}
|
|
3650
|
+
if (performance.avgCompleteness < 70) {
|
|
3651
|
+
return {
|
|
3652
|
+
trigger: true,
|
|
3653
|
+
reason: `Low completeness (${performance.avgCompleteness.toFixed(1)})`
|
|
3654
|
+
};
|
|
3655
|
+
}
|
|
3656
|
+
if (performance.stdDevCompleteness > 25) {
|
|
3657
|
+
return {
|
|
3658
|
+
trigger: true,
|
|
3659
|
+
reason: "High performance variability detected"
|
|
3660
|
+
};
|
|
3661
|
+
}
|
|
3662
|
+
return {
|
|
3663
|
+
trigger: false,
|
|
3664
|
+
reason: "Performance meets standards"
|
|
3665
|
+
};
|
|
3666
|
+
}
|
|
3667
|
+
};
|
|
3668
|
+
|
|
3669
|
+
export {
|
|
3670
|
+
aggregateScores,
|
|
3671
|
+
buildCriterionScores,
|
|
3672
|
+
ProviderTracker,
|
|
3673
|
+
TunerActivationAuthority,
|
|
3674
|
+
PromptEvaluator,
|
|
3675
|
+
PromptOptimizer,
|
|
3676
|
+
AutoTuningEngine,
|
|
3677
|
+
AutomaticPromptOptimizer
|
|
3678
|
+
};
|
|
3679
|
+
//# sourceMappingURL=chunk-7MMLMXO6.js.map
|