@poetic-ai/poetic 1.42.1-bootstrap.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/SECURITY.md +47 -0
- package/.nvmrc +1 -0
- package/.poetic/README.md +37 -0
- package/.poetic/providers/catalog.json +12051 -0
- package/.poetic/providers/pricing.json +1272 -0
- package/.poetic/providers/registry.json +3369 -0
- package/CHANGELOG.md +552 -0
- package/CODE_OF_CONDUCT.md +40 -0
- package/CONTRIBUTING.md +23 -0
- package/INSTALL.md +454 -0
- package/LICENSE +21 -0
- package/README.md +474 -0
- package/dist/BasicOptimizationCompetitionRunner-5H5TBYO4.js +153 -0
- package/dist/ExecutionTracker-UVONC4C6.js +16 -0
- package/dist/OptimizationConfig-7DFY2TST.js +17 -0
- package/dist/OptimizationEngine-MXTSOSST.js +1597 -0
- package/dist/PromptStore-U6ZD3ZMQ.js +18 -0
- package/dist/actions-M3AESHFF.js +322 -0
- package/dist/agent-ingest-HO77TH26.js +53 -0
- package/dist/aggregator-DNCBINFO.js +12 -0
- package/dist/ai-judge-LDWHURKK.js +156 -0
- package/dist/allowlist-grounding-RO6HKUFI.js +16 -0
- package/dist/allowlist-utils-23V7TJ3G.js +28 -0
- package/dist/anchored-turn-service-GFYU6GCQ.js +489 -0
- package/dist/api-transport-FS3CQRMN.js +1087 -0
- package/dist/apply-completion-mode-JBT2CTQO.js +89 -0
- package/dist/artifact-migrator-23L45ISD.js +267 -0
- package/dist/ask-UYVZY7JH.js +135 -0
- package/dist/auth-CSQP5RHJ.js +67 -0
- package/dist/auth-E6XUNZ5M.js +68 -0
- package/dist/auth-KQLJPZ2E.js +134 -0
- package/dist/auth-OC4E6633.js +71 -0
- package/dist/auth-R3NQ2ZF5.js +78 -0
- package/dist/auth-S4CTX4GQ.js +44 -0
- package/dist/auth-YLH3S7EY.js +69 -0
- package/dist/auth-liveness-TDDEZY3T.js +46 -0
- package/dist/auto-optimizer-HA3T23SF.js +80 -0
- package/dist/autoloop-cleanup-JOGKTLWA.js +109 -0
- package/dist/autoloop-evidence-gates-ACB5YW5G.js +36 -0
- package/dist/autoloop-gate-policy-XZA5EQFN.js +175 -0
- package/dist/autoloop-helpers-EFJLBQUP.js +22 -0
- package/dist/autoloop-ledger-W7QO62PX.js +96 -0
- package/dist/autoloop-ledger-subscriber-Z5RW5EIE.js +162 -0
- package/dist/autoloop-run-defaults-FERGRZGP.js +19 -0
- package/dist/autoloop-run-lock-CT2NWSV3.js +197 -0
- package/dist/autoloop-success-OPR74YEX.js +85 -0
- package/dist/autoloop-termination-summary-RN75F422.js +58 -0
- package/dist/autoloop-trajectory-K5I45UBR.js +282 -0
- package/dist/autoloop-wall-clock-T7ZVIQOC.js +12 -0
- package/dist/autonomous-loop-controller-4QJMOIIN.js +4120 -0
- package/dist/backend-P3DAYRP7.js +29 -0
- package/dist/background-executor-CBN552V4.js +257 -0
- package/dist/backlog-execution-intent-U4XYPTV6.js +15 -0
- package/dist/backup-active-store-5WXG6SXT.js +17 -0
- package/dist/backup-telemetry-db-XGLXAA3N.js +254 -0
- package/dist/basic-competition-master-WXZHUT42.js +344 -0
- package/dist/branch-archive-manager-RP6FP3ZC.js +254 -0
- package/dist/branch-cleanup-manager-RSHAI7YR.js +20 -0
- package/dist/build-gate-YFASRB5Z.js +71 -0
- package/dist/change-summary-WY7J5TOX.js +33 -0
- package/dist/chunk-25QL3M3L.js +899 -0
- package/dist/chunk-2ABC6SEC.js +2626 -0
- package/dist/chunk-2AICM4P3.js +1763 -0
- package/dist/chunk-2AQUWJPW.js +225 -0
- package/dist/chunk-2DY5KNVM.js +276 -0
- package/dist/chunk-2QJ3H3L7.js +42 -0
- package/dist/chunk-2UH6VQRZ.js +328 -0
- package/dist/chunk-2VSQMRCB.js +18 -0
- package/dist/chunk-2XQXYBM2.js +130 -0
- package/dist/chunk-2YFUIER7.js +426 -0
- package/dist/chunk-2YXZCZTX.js +1160 -0
- package/dist/chunk-32VLDFRT.js +797 -0
- package/dist/chunk-34CNP2HB.js +148 -0
- package/dist/chunk-35P3W3JX.js +456 -0
- package/dist/chunk-36PEBZAF.js +18 -0
- package/dist/chunk-3733QTI7.js +6818 -0
- package/dist/chunk-3A5PDH5L.js +12513 -0
- package/dist/chunk-3AOKYKA7.js +22 -0
- package/dist/chunk-3BCUNHKF.js +475 -0
- package/dist/chunk-3C26DPFK.js +566 -0
- package/dist/chunk-3FLXLLB7.js +26 -0
- package/dist/chunk-3FOZ2VHV.js +100 -0
- package/dist/chunk-3G6FZCRF.js +106 -0
- package/dist/chunk-3JWNRJSB.js +736 -0
- package/dist/chunk-3LUMR3D3.js +156 -0
- package/dist/chunk-3M57DADF.js +44 -0
- package/dist/chunk-3S62LLWJ.js +543 -0
- package/dist/chunk-3VABAKT2.js +1784 -0
- package/dist/chunk-44N5WZZH.js +45 -0
- package/dist/chunk-45FPSA3B.js +125 -0
- package/dist/chunk-4BIPZBT7.js +791 -0
- package/dist/chunk-4BJ3O4RQ.js +929 -0
- package/dist/chunk-4BTDUT2S.js +42 -0
- package/dist/chunk-4CSS7WVI.js +143 -0
- package/dist/chunk-4CUY7AFY.js +247 -0
- package/dist/chunk-4HN6DF7V.js +1027 -0
- package/dist/chunk-4HXKHDNH.js +1064 -0
- package/dist/chunk-4KKI477Q.js +44 -0
- package/dist/chunk-4NA7FSTV.js +595 -0
- package/dist/chunk-4NXPXE62.js +84 -0
- package/dist/chunk-4SALLK63.js +9831 -0
- package/dist/chunk-4SHKPCGK.js +110 -0
- package/dist/chunk-4VJITV4P.js +445 -0
- package/dist/chunk-4W2NRXQL.js +1205 -0
- package/dist/chunk-4XICK3HN.js +784 -0
- package/dist/chunk-55EJVV3C.js +385 -0
- package/dist/chunk-5DLKYTQX.js +4003 -0
- package/dist/chunk-5EERLVZB.js +139 -0
- package/dist/chunk-5HDP7XZG.js +91 -0
- package/dist/chunk-5LAMRH4X.js +969 -0
- package/dist/chunk-5REDMNLZ.js +310 -0
- package/dist/chunk-5UCJARII.js +1984 -0
- package/dist/chunk-5WCQRDRB.js +588 -0
- package/dist/chunk-5WRLK5KW.js +700 -0
- package/dist/chunk-66AGFTBQ.js +131 -0
- package/dist/chunk-66CIO2SV.js +48 -0
- package/dist/chunk-6BM72ZMG.js +31 -0
- package/dist/chunk-6BTFXYRO.js +276 -0
- package/dist/chunk-6EDAJH2I.js +68 -0
- package/dist/chunk-6EQKFDBH.js +457 -0
- package/dist/chunk-6KMJKHIQ.js +2193 -0
- package/dist/chunk-6MFRHTRI.js +115 -0
- package/dist/chunk-6QMUKUHH.js +632 -0
- package/dist/chunk-6TZJRKNW.js +321 -0
- package/dist/chunk-6V7HGQFZ.js +20 -0
- package/dist/chunk-6Y5TWI7H.js +146 -0
- package/dist/chunk-6Y5U7UFY.js +91 -0
- package/dist/chunk-6ZLAK3XG.js +14 -0
- package/dist/chunk-72XCRQED.js +34 -0
- package/dist/chunk-72YTZMAM.js +589 -0
- package/dist/chunk-73NCTWKL.js +188 -0
- package/dist/chunk-77I6G4CO.js +1 -0
- package/dist/chunk-7D7QT3BE.js +688 -0
- package/dist/chunk-7DNSNKJV.js +78 -0
- package/dist/chunk-7FMJVYBS.js +307 -0
- package/dist/chunk-7GDRQBQC.js +419 -0
- package/dist/chunk-7IL4A2PP.js +948 -0
- package/dist/chunk-7NPSXSHO.js +975 -0
- package/dist/chunk-7PS6INZP.js +302 -0
- package/dist/chunk-7QCZSLBH.js +1 -0
- package/dist/chunk-7R2WLDOS.js +62 -0
- package/dist/chunk-7XV3QXLT.js +878 -0
- package/dist/chunk-A2WKSVX7.js +43 -0
- package/dist/chunk-A3YGID55.js +133 -0
- package/dist/chunk-A5MRWHMI.js +1604 -0
- package/dist/chunk-A722DEVA.js +14 -0
- package/dist/chunk-AG2Z2SUZ.js +554 -0
- package/dist/chunk-AHWAZ2MU.js +137 -0
- package/dist/chunk-AIAQ3HS3.js +119 -0
- package/dist/chunk-ALAB3PBR.js +2206 -0
- package/dist/chunk-APPNGGN7.js +365 -0
- package/dist/chunk-APV6MK5E.js +137 -0
- package/dist/chunk-AQNPGRWS.js +395 -0
- package/dist/chunk-ASEP3M2W.js +968 -0
- package/dist/chunk-AT4TNPWW.js +2931 -0
- package/dist/chunk-ATVKCSGT.js +250 -0
- package/dist/chunk-AX6GWE7V.js +47 -0
- package/dist/chunk-AZQILSNQ.js +1376 -0
- package/dist/chunk-B76KHLX6.js +13 -0
- package/dist/chunk-BAV2HJXS.js +194 -0
- package/dist/chunk-BBHRN366.js +89 -0
- package/dist/chunk-BCZAENOH.js +448 -0
- package/dist/chunk-BIGSACCY.js +19527 -0
- package/dist/chunk-BPYMCIVD.js +9243 -0
- package/dist/chunk-BSE4R6XD.js +322 -0
- package/dist/chunk-BVPWRDCO.js +2321 -0
- package/dist/chunk-BXI4NXL3.js +507 -0
- package/dist/chunk-C3L6YQ7P.js +147 -0
- package/dist/chunk-C57KJOUQ.js +121 -0
- package/dist/chunk-CCV7BPSY.js +118 -0
- package/dist/chunk-CD7ISXA4.js +826 -0
- package/dist/chunk-CE5OZBYY.js +21 -0
- package/dist/chunk-CFBIG37O.js +154 -0
- package/dist/chunk-CHQV73H5.js +408 -0
- package/dist/chunk-CJC6GZ46.js +457 -0
- package/dist/chunk-CJHRFSNY.js +924 -0
- package/dist/chunk-CKP2Q3TP.js +1905 -0
- package/dist/chunk-CLKSETWE.js +10 -0
- package/dist/chunk-CQM3A35X.js +844 -0
- package/dist/chunk-CSLAS5GW.js +4007 -0
- package/dist/chunk-CTLUBNCW.js +622 -0
- package/dist/chunk-CVMVZWG4.js +3544 -0
- package/dist/chunk-CYB6QDCT.js +263 -0
- package/dist/chunk-D5EP5D2W.js +26 -0
- package/dist/chunk-DC4ZYMXJ.js +115 -0
- package/dist/chunk-DCAVAABI.js +348 -0
- package/dist/chunk-DCYF7EKG.js +26 -0
- package/dist/chunk-DF5SLDE4.js +966 -0
- package/dist/chunk-DN7UMLF7.js +5678 -0
- package/dist/chunk-DSFM56OA.js +1203 -0
- package/dist/chunk-DTY4APYV.js +218 -0
- package/dist/chunk-E3NTPHEO.js +194 -0
- package/dist/chunk-E74LUIID.js +302 -0
- package/dist/chunk-EJ42BNGY.js +2904 -0
- package/dist/chunk-EJ4ZAWLV.js +305 -0
- package/dist/chunk-EJGCZGUF.js +1483 -0
- package/dist/chunk-EK2U3YEG.js +486 -0
- package/dist/chunk-ELMFYNV7.js +109 -0
- package/dist/chunk-ENRGKNR5.js +25 -0
- package/dist/chunk-EQGIDAUF.js +62 -0
- package/dist/chunk-EX3OM35G.js +20414 -0
- package/dist/chunk-F5JO7HAS.js +94 -0
- package/dist/chunk-F7ER6ANO.js +174 -0
- package/dist/chunk-FATUVIIN.js +374 -0
- package/dist/chunk-FC23CTXV.js +322 -0
- package/dist/chunk-FCNAS3W2.js +1033 -0
- package/dist/chunk-FD4ERYEG.js +22 -0
- package/dist/chunk-FDVZLG6Z.js +3104 -0
- package/dist/chunk-FEPIFY7D.js +511 -0
- package/dist/chunk-FEQ4HXL7.js +412 -0
- package/dist/chunk-FFH5HHJU.js +5351 -0
- package/dist/chunk-FH7WOJAE.js +79 -0
- package/dist/chunk-FKAUPHMR.js +96 -0
- package/dist/chunk-FRG57FPM.js +240 -0
- package/dist/chunk-FUVP64ER.js +31 -0
- package/dist/chunk-G2PTLLEL.js +284 -0
- package/dist/chunk-G35UYQ6Z.js +69 -0
- package/dist/chunk-G36W2T2S.js +1143 -0
- package/dist/chunk-G4THLFV3.js +148 -0
- package/dist/chunk-GAFJEYLK.js +29 -0
- package/dist/chunk-GCCAPO3S.js +118 -0
- package/dist/chunk-GG6KJYW6.js +85 -0
- package/dist/chunk-GHHFDU2U.js +55 -0
- package/dist/chunk-GHTZSQB7.js +57 -0
- package/dist/chunk-GJTWJNLV.js +1374 -0
- package/dist/chunk-GKN3HMN4.js +2809 -0
- package/dist/chunk-GNH3QGTL.js +1342 -0
- package/dist/chunk-GOAQ2I7Y.js +7000 -0
- package/dist/chunk-GSGD5E4C.js +30 -0
- package/dist/chunk-GXWAJQOJ.js +5011 -0
- package/dist/chunk-H2AYQIHW.js +224 -0
- package/dist/chunk-H6G7QRMQ.js +35 -0
- package/dist/chunk-H7JAGHZT.js +59 -0
- package/dist/chunk-HAC6EHZZ.js +79 -0
- package/dist/chunk-HE2KH6EQ.js +161 -0
- package/dist/chunk-HIHENQDX.js +66 -0
- package/dist/chunk-HIO33P3G.js +19 -0
- package/dist/chunk-HK4AQZSR.js +438 -0
- package/dist/chunk-HMVQIPH3.js +547 -0
- package/dist/chunk-HOYBYAOA.js +257 -0
- package/dist/chunk-HQLGXODD.js +123 -0
- package/dist/chunk-HQXYYUNY.js +1231 -0
- package/dist/chunk-HY3ESWCA.js +708 -0
- package/dist/chunk-I46EG2XQ.js +65 -0
- package/dist/chunk-I4NPDSCB.js +279 -0
- package/dist/chunk-I5WRT7ZR.js +514 -0
- package/dist/chunk-I6BL3PDH.js +1024 -0
- package/dist/chunk-IATQ7NBI.js +679 -0
- package/dist/chunk-IC7I6YIJ.js +301 -0
- package/dist/chunk-IDCSOZVN.js +78 -0
- package/dist/chunk-IEXU4RR6.js +5742 -0
- package/dist/chunk-IFCBYECK.js +112 -0
- package/dist/chunk-IFZCPEY2.js +66 -0
- package/dist/chunk-IKFP2KUR.js +48 -0
- package/dist/chunk-IP7BYVUV.js +338 -0
- package/dist/chunk-IRJ2I6KH.js +166 -0
- package/dist/chunk-IUXNGKA6.js +152 -0
- package/dist/chunk-IZYIQQ7X.js +53 -0
- package/dist/chunk-IZZK3H6I.js +2175 -0
- package/dist/chunk-J3J4P7RW.js +353 -0
- package/dist/chunk-J62KPGII.js +464 -0
- package/dist/chunk-J6P3FWWW.js +130 -0
- package/dist/chunk-JD6SIZED.js +712 -0
- package/dist/chunk-JEV6UI7H.js +537 -0
- package/dist/chunk-JKBHJQMT.js +434 -0
- package/dist/chunk-JLS65NUQ.js +14 -0
- package/dist/chunk-JOSYEJKY.js +141 -0
- package/dist/chunk-JPKSXKNB.js +94 -0
- package/dist/chunk-JQD5VYKR.js +483 -0
- package/dist/chunk-JSOWEFEQ.js +351 -0
- package/dist/chunk-JSWMRQYG.js +28 -0
- package/dist/chunk-K4XK7SGB.js +6783 -0
- package/dist/chunk-K54Y7MLK.js +160 -0
- package/dist/chunk-KAT4RXUT.js +521 -0
- package/dist/chunk-KDMKKIRK.js +103 -0
- package/dist/chunk-KM3F7SYE.js +702 -0
- package/dist/chunk-KNDYMVWN.js +102 -0
- package/dist/chunk-KNT72VR6.js +132 -0
- package/dist/chunk-KQEBSK3T.js +274 -0
- package/dist/chunk-KT6VHSPF.js +84 -0
- package/dist/chunk-KTMW65UX.js +14 -0
- package/dist/chunk-L4427KOE.js +26 -0
- package/dist/chunk-L6OM4A22.js +189 -0
- package/dist/chunk-L7E6XCRA.js +1042 -0
- package/dist/chunk-L7JZLXQL.js +173 -0
- package/dist/chunk-LAMKBTXX.js +40 -0
- package/dist/chunk-LDCF2MKN.js +132 -0
- package/dist/chunk-LEIIWPWQ.js +78 -0
- package/dist/chunk-LFTQVYBB.js +31 -0
- package/dist/chunk-LGHII5ET.js +23 -0
- package/dist/chunk-LH34OGSB.js +265 -0
- package/dist/chunk-LO3X5NLS.js +326 -0
- package/dist/chunk-LPITVM7M.js +339 -0
- package/dist/chunk-LR5IJIGV.js +810 -0
- package/dist/chunk-LUVKFFWR.js +315 -0
- package/dist/chunk-LWCBNGH6.js +35 -0
- package/dist/chunk-LWTN3WRV.js +260 -0
- package/dist/chunk-LZBBXYWF.js +1004 -0
- package/dist/chunk-M34PS4NE.js +2213 -0
- package/dist/chunk-M3HQYKQX.js +308 -0
- package/dist/chunk-MCPQNJMK.js +482 -0
- package/dist/chunk-MF5HP6XV.js +221 -0
- package/dist/chunk-MFADRRTR.js +112 -0
- package/dist/chunk-MN6PL4AN.js +120 -0
- package/dist/chunk-MQ4WA34C.js +206 -0
- package/dist/chunk-MTVZITQ2.js +185 -0
- package/dist/chunk-MWHF5V7U.js +223 -0
- package/dist/chunk-MWHLBPNU.js +211 -0
- package/dist/chunk-MWK5UZQ3.js +1068 -0
- package/dist/chunk-MYLH2S4G.js +64 -0
- package/dist/chunk-MZ5WYKNA.js +28 -0
- package/dist/chunk-N5DX4JAK.js +12 -0
- package/dist/chunk-NEDCMN7E.js +415 -0
- package/dist/chunk-NEQUQHOX.js +223 -0
- package/dist/chunk-NGK6PEOG.js +25 -0
- package/dist/chunk-NM6PMYY4.js +768 -0
- package/dist/chunk-NOZGR2QO.js +2373 -0
- package/dist/chunk-NP2V3L7K.js +226 -0
- package/dist/chunk-NPZJMUZ4.js +194 -0
- package/dist/chunk-NRLY536M.js +260 -0
- package/dist/chunk-NS33V3IM.js +51 -0
- package/dist/chunk-NWOYOA6H.js +378 -0
- package/dist/chunk-NWRUSEXG.js +322 -0
- package/dist/chunk-O6V4PNLO.js +63 -0
- package/dist/chunk-OGMSYL6J.js +227 -0
- package/dist/chunk-OIS7J26S.js +319 -0
- package/dist/chunk-OKBTRSZI.js +205 -0
- package/dist/chunk-OMX5VL43.js +79 -0
- package/dist/chunk-ON3G73BU.js +264 -0
- package/dist/chunk-OPYFYTWQ.js +220 -0
- package/dist/chunk-OQ6K46CK.js +1734 -0
- package/dist/chunk-OSDYVKHW.js +2440 -0
- package/dist/chunk-OUM5S64R.js +3429 -0
- package/dist/chunk-OWB7JXUS.js +751 -0
- package/dist/chunk-OYZYO7TV.js +47 -0
- package/dist/chunk-P47VN5F4.js +282 -0
- package/dist/chunk-P6VQROCO.js +594 -0
- package/dist/chunk-P7CQGPLS.js +30 -0
- package/dist/chunk-PA6BCSOX.js +2872 -0
- package/dist/chunk-PGSPX4SU.js +15 -0
- package/dist/chunk-PIKLF7BM.js +432 -0
- package/dist/chunk-PJWJ3SAV.js +22 -0
- package/dist/chunk-PKIFMV72.js +100 -0
- package/dist/chunk-PTH5E5XO.js +195 -0
- package/dist/chunk-PYITM4N2.js +79 -0
- package/dist/chunk-PZ5AY32C.js +10 -0
- package/dist/chunk-Q3RO7N35.js +45 -0
- package/dist/chunk-Q3WIC6GQ.js +2670 -0
- package/dist/chunk-QGJDMEOV.js +34 -0
- package/dist/chunk-QJYIHVXT.js +100 -0
- package/dist/chunk-QNOB37UH.js +58 -0
- package/dist/chunk-QOCOVLUH.js +462 -0
- package/dist/chunk-QTCJ5CC5.js +172 -0
- package/dist/chunk-QU6JQJMH.js +3063 -0
- package/dist/chunk-QVFV5IP2.js +232 -0
- package/dist/chunk-QVZMFDYG.js +29 -0
- package/dist/chunk-R2ZYWNW3.js +1581 -0
- package/dist/chunk-R4UWBC35.js +200 -0
- package/dist/chunk-R6W23LKN.js +254 -0
- package/dist/chunk-RAK5HOIJ.js +459 -0
- package/dist/chunk-RDFRCT64.js +168 -0
- package/dist/chunk-RE5VZDFQ.js +58 -0
- package/dist/chunk-RGROYC2G.js +638 -0
- package/dist/chunk-RGWHUI5E.js +68 -0
- package/dist/chunk-RHVG6UNL.js +205 -0
- package/dist/chunk-RI4EX2QE.js +18 -0
- package/dist/chunk-RIKFAWZF.js +71 -0
- package/dist/chunk-RJAAKDRX.js +17 -0
- package/dist/chunk-RSXSD4MH.js +1016 -0
- package/dist/chunk-RW7RVSDV.js +79 -0
- package/dist/chunk-RWREC3KJ.js +517 -0
- package/dist/chunk-RZY7RKM5.js +500 -0
- package/dist/chunk-S2MNWFAG.js +52 -0
- package/dist/chunk-S2VQCZO4.js +22 -0
- package/dist/chunk-S54MKU6V.js +117 -0
- package/dist/chunk-S77XPALC.js +239 -0
- package/dist/chunk-SCW4ZF6R.js +150 -0
- package/dist/chunk-SKHGYQ7W.js +405 -0
- package/dist/chunk-SNLBO4BF.js +247 -0
- package/dist/chunk-SP3LFU6B.js +60 -0
- package/dist/chunk-SQZRPY23.js +5181 -0
- package/dist/chunk-SSB5UQHA.js +4128 -0
- package/dist/chunk-STJOHHSF.js +605 -0
- package/dist/chunk-STV6LYYE.js +45 -0
- package/dist/chunk-SUN4OFTW.js +740 -0
- package/dist/chunk-SURZ2WFE.js +326 -0
- package/dist/chunk-SUZ7UYZH.js +1972 -0
- package/dist/chunk-SZ7DL357.js +1047 -0
- package/dist/chunk-T2SZELJL.js +624 -0
- package/dist/chunk-T6VPI2XP.js +10338 -0
- package/dist/chunk-TAJPCXSB.js +13 -0
- package/dist/chunk-TAOOYK3P.js +358 -0
- package/dist/chunk-TDSEX5CF.js +24 -0
- package/dist/chunk-TFTGP7Z7.js +174 -0
- package/dist/chunk-TGON4N4O.js +197 -0
- package/dist/chunk-TIBZPAD2.js +9924 -0
- package/dist/chunk-TK644IOG.js +1813 -0
- package/dist/chunk-TRSVUPCX.js +25 -0
- package/dist/chunk-TSPQONRT.js +4533 -0
- package/dist/chunk-TSXGRLPR.js +107 -0
- package/dist/chunk-TWYLVAS2.js +152 -0
- package/dist/chunk-U2S3YEX3.js +224 -0
- package/dist/chunk-U32HWI76.js +437 -0
- package/dist/chunk-U45RTRGY.js +818 -0
- package/dist/chunk-U62VCQGR.js +49 -0
- package/dist/chunk-UDMS5AD3.js +307 -0
- package/dist/chunk-UGX7GI37.js +94 -0
- package/dist/chunk-UH3JJSHK.js +1 -0
- package/dist/chunk-UHRBTZYY.js +208 -0
- package/dist/chunk-UHT2KRG5.js +193 -0
- package/dist/chunk-UI2F6DJ5.js +118 -0
- package/dist/chunk-UKMAFG7A.js +189 -0
- package/dist/chunk-UUH2RVKS.js +585 -0
- package/dist/chunk-UVBBFDFQ.js +270 -0
- package/dist/chunk-UXZF3B7T.js +75 -0
- package/dist/chunk-V6YLL5WA.js +3746 -0
- package/dist/chunk-V72RMBM4.js +48 -0
- package/dist/chunk-VBNCKCCI.js +16 -0
- package/dist/chunk-VCAUV5XQ.js +3679 -0
- package/dist/chunk-VDSH7O4Y.js +152 -0
- package/dist/chunk-VNSYKC25.js +24 -0
- package/dist/chunk-VTLNJQ44.js +133 -0
- package/dist/chunk-VV6HOXRC.js +302 -0
- package/dist/chunk-VVGEPBPS.js +218 -0
- package/dist/chunk-VXF4Q3FW.js +20 -0
- package/dist/chunk-W3ZGW3C5.js +331 -0
- package/dist/chunk-W4OMESPG.js +271 -0
- package/dist/chunk-WDUORIHF.js +877 -0
- package/dist/chunk-WDVTY4U6.js +148 -0
- package/dist/chunk-WGXYDLMX.js +4386 -0
- package/dist/chunk-WMPU2UOV.js +82 -0
- package/dist/chunk-WOEVVPDP.js +1125 -0
- package/dist/chunk-WTLMSPBQ.js +256 -0
- package/dist/chunk-X43NMU4Z.js +1124 -0
- package/dist/chunk-X5NTI5U5.js +368 -0
- package/dist/chunk-X6UIEADS.js +949 -0
- package/dist/chunk-XAONGNST.js +57 -0
- package/dist/chunk-XB5YLCLB.js +13 -0
- package/dist/chunk-XGYF2QMZ.js +351 -0
- package/dist/chunk-XHEKQENB.js +255 -0
- package/dist/chunk-XPHKYHO4.js +878 -0
- package/dist/chunk-XQD2B6BB.js +507 -0
- package/dist/chunk-XT2ZQVTC.js +558 -0
- package/dist/chunk-XUSRFFA7.js +1859 -0
- package/dist/chunk-XUXVDPSZ.js +50 -0
- package/dist/chunk-XV7NZL4B.js +57 -0
- package/dist/chunk-XXCGJGSQ.js +121 -0
- package/dist/chunk-XYGDZCB5.js +659 -0
- package/dist/chunk-XZAMZUX5.js +992 -0
- package/dist/chunk-XZE4XARH.js +563 -0
- package/dist/chunk-Y34DUC4A.js +1356 -0
- package/dist/chunk-Y3J4JMTU.js +61 -0
- package/dist/chunk-Y4RC2YFX.js +1270 -0
- package/dist/chunk-Y55LZN5U.js +52 -0
- package/dist/chunk-Y5677NDO.js +183 -0
- package/dist/chunk-YESZJFK6.js +220 -0
- package/dist/chunk-YGXKOBQQ.js +292 -0
- package/dist/chunk-YMX76IOS.js +31 -0
- package/dist/chunk-YOTXESEH.js +257 -0
- package/dist/chunk-YQAAKTVH.js +712 -0
- package/dist/chunk-YVORHQ2S.js +579 -0
- package/dist/chunk-YWNAHW24.js +1552 -0
- package/dist/chunk-Z476RATD.js +1689 -0
- package/dist/chunk-ZDTGECTN.js +235 -0
- package/dist/chunk-ZDTMICNY.js +316 -0
- package/dist/chunk-ZIPWI2LY.js +151 -0
- package/dist/chunk-ZIVYXF7A.js +2344 -0
- package/dist/chunk-ZX65UI5W.js +78 -0
- package/dist/chunk-ZYG7GGKO.js +218 -0
- package/dist/chunk-ZZZUL4OF.js +38 -0
- package/dist/cli-utils-HUOTBRM3.js +40 -0
- package/dist/cli-validation-WLXWLUJL.js +111 -0
- package/dist/commit-utils-3QV2P6RA.js +202 -0
- package/dist/compete-config-resolver-SNH3TT2Q.js +76 -0
- package/dist/compete-request-XWNPOARL.js +364 -0
- package/dist/compete-results-PAP67LDR.js +208 -0
- package/dist/competition-outcome-tracker-GOZXNAQU.js +38 -0
- package/dist/config-HWHUNA72.js +166 -0
- package/dist/config-audit-EL3GHXS7.js +360 -0
- package/dist/config-explain-TNILDEVJ.js +12 -0
- package/dist/config-loader-VPFPFO4B.js +38 -0
- package/dist/config-manager-L3VRNTSG.js +67 -0
- package/dist/config-validator-SIIQEOFQ.js +72 -0
- package/dist/cost-7CX37DU3.js +53 -0
- package/dist/coverage-orchestrator-MNKHD6YC.js +772 -0
- package/dist/coverage-scanner-YVD62MLH.js +12 -0
- package/dist/createCompetitionRunner-LJG6ZQSO.js +46 -0
- package/dist/data-migration-state-VJK64DLT.js +31 -0
- package/dist/db-migrator-XSEIOGWY.js +34 -0
- package/dist/discovery-JGYGDD25.js +25 -0
- package/dist/doctor-CGBCD7IQ.js +186 -0
- package/dist/domain-analyzer-DL6BDR2N.js +29 -0
- package/dist/ensure-initialized-4BY73HLE.js +52 -0
- package/dist/entry.js +940 -0
- package/dist/environment-IGVH5X53.js +81 -0
- package/dist/error-parser-core-PPZ36A4S.js +50 -0
- package/dist/escalation-ladder-NWY5BJ57.js +130 -0
- package/dist/evaluation-context-KMWDSDYD.js +35 -0
- package/dist/evidence-plan-resolver-YIVH76PU.js +85 -0
- package/dist/execution-data-writer-OHAQ7G5H.js +53 -0
- package/dist/execution-policy-applier-ZHD3DCF3.js +31 -0
- package/dist/execution-preflight-Z4Y64V3I.js +209 -0
- package/dist/execution-roles-KBH2MC3M.js +66 -0
- package/dist/exit-code-error-Z4SW2DVK.js +14 -0
- package/dist/external-temp-cleanup-N2RZH4QP.js +311 -0
- package/dist/factory-IYBMCZG7.js +149 -0
- package/dist/file-utils-XIVR2ZMA.js +47 -0
- package/dist/flywheel-autoloop-executor-R3KD2G4D.js +249 -0
- package/dist/flywheel-competition-executor-XCFGVJSK.js +259 -0
- package/dist/flywheel-executor-LTSIZN2L.js +369 -0
- package/dist/flywheel-git-isolation-CZ5YVVZY.js +48 -0
- package/dist/flywheel-manifest-5UC4VBMA.js +108 -0
- package/dist/flywheel-model-defaults-7CVKP53V.js +27 -0
- package/dist/flywheel-preflight-6E33DJDK.js +104 -0
- package/dist/flywheel-resume-7HELH7QS.js +23 -0
- package/dist/flywheel-safety-7U7DPOAP.js +28 -0
- package/dist/flywheel-scope-decision-ZHKF6ZIN.js +11 -0
- package/dist/get-telemetry-logger-AZ46EPRP.js +73 -0
- package/dist/git-worktree-IY6V6DUO.js +18 -0
- package/dist/github-pr-manager-4XUO2YTM.js +211 -0
- package/dist/guidance-profile-config-E7XJL2VP.js +117 -0
- package/dist/guidance-profile-renderer-QK2YMAAP.js +21 -0
- package/dist/index-query-4JEUKA44.js +31 -0
- package/dist/index.js +108644 -0
- package/dist/instructions-CM6THDV7.js +64 -0
- package/dist/invoke-provider-auth-V5TC5UDS.js +223 -0
- package/dist/judge-scoring.json +80 -0
- package/dist/judge-test.js +555 -0
- package/dist/lab-mode-OTWVHN63.js +39 -0
- package/dist/lab-utils-VA6VXNOV.js +23 -0
- package/dist/linux-host-class-UY4KVLFD.js +29 -0
- package/dist/llm-judge-executor-34YPIRJ7.js +133 -0
- package/dist/local-executor-LJJSCWZX.js +223 -0
- package/dist/logger-6PRYM56M.js +72 -0
- package/dist/loop-spec-XTLQAVZ7.js +191 -0
- package/dist/manager-2HJG5CLT.js +25 -0
- package/dist/matrix-prompt-builder-LHDTJ4TY.js +497 -0
- package/dist/matrix-result-parser-7W35EUOV.js +15 -0
- package/dist/metadata-HQ43NDEK.js +22 -0
- package/dist/model-invocations-db-QBO4Y7WZ.js +38 -0
- package/dist/monitor-MCI3F4PX.js +72 -0
- package/dist/next-HQKDXIIR.js +90 -0
- package/dist/next-JC3V2UCS.js +69 -0
- package/dist/optimization-config-resolver-QPSFLCJU.js +75 -0
- package/dist/package-3YCWIQQ5.js +10 -0
- package/dist/parallel-orchestrator-NL7YJJU2.js +297 -0
- package/dist/parallel-worker.js +279 -0
- package/dist/parse-git-status-PIAF3OTP.js +10 -0
- package/dist/path-security-ETTH6TZL.js +39 -0
- package/dist/pattern-bank-UNJR5ADV.js +15 -0
- package/dist/plan-backlog-list-fast-DZK24LSZ.js +270 -0
- package/dist/plan-backlog-show-fast-63GBBH6Q.js +406 -0
- package/dist/plan-sprint-list-fast-TJ6ZUF5E.js +162 -0
- package/dist/plan-sprint-status-fast-YWE5X3NG.js +208 -0
- package/dist/plan-task-status-fast-K22BTTBW.js +349 -0
- package/dist/plan-to-flywheel-VFGCPINA.js +456 -0
- package/dist/planning-backlog-Y6TIJYWU.js +319 -0
- package/dist/poetic-root-V5DNXEAE.js +17 -0
- package/dist/preflight-OPANWMME.js +12 -0
- package/dist/process-registry-PDZJPTAS.js +27 -0
- package/dist/process-scanner-C22SYXYD.js +22 -0
- package/dist/processes-KMQUD7HY.js +29 -0
- package/dist/processes-json-fast-IBDXZ6WW.js +135 -0
- package/dist/profile-2XP3FWH2.js +149 -0
- package/dist/prompts-EPAAAMMF.js +168 -0
- package/dist/protected-branches-RNTEX7ET.js +57 -0
- package/dist/provider-aware-resource-manager-QBTZJMNP.js +18 -0
- package/dist/provider-matrix-KWTYA5QV.js +154 -0
- package/dist/provider-matrix-renderer-MZDWKKID.js +139 -0
- package/dist/provider-output-forwarding-DERID3TP.js +30 -0
- package/dist/provider-registry-J5NRBJKI.js +128 -0
- package/dist/provider-temp-cleanup-KYGLF6YU.js +14 -0
- package/dist/prune-engine-Q6E2DXBN.js +451 -0
- package/dist/quality-gate-33SHK43X.js +70 -0
- package/dist/quickstart-E32SODTA.js +62 -0
- package/dist/readiness-REWW7OSV.js +96 -0
- package/dist/registry-N6L36RE2.js +153 -0
- package/dist/repair-helpers-TTV5FEF4.js +117 -0
- package/dist/repo-root-2IGD4H7X.js +22 -0
- package/dist/resolution-engine-LXQYDSKS.js +127 -0
- package/dist/resolve-provider-cli-O2P6XZBZ.js +35 -0
- package/dist/restore-telemetry-db-BQL6DTQF.js +360 -0
- package/dist/result-streamer-WEHW476R.js +19 -0
- package/dist/routing-events-DHPUZDDF.js +27 -0
- package/dist/routing-history-store-UWCEXAP3.js +46 -0
- package/dist/routing-planner-VGQU35QM.js +160 -0
- package/dist/run-poetic-tui-ZZXFDOQ4.js +13517 -0
- package/dist/run-simulation.js +376 -0
- package/dist/runner-2YPKOJ6A.js +1092 -0
- package/dist/runner-HEJJ65AJ.js +1571 -0
- package/dist/safety-OCC4KJJR.js +64 -0
- package/dist/safety-stanza-BSVE6OJU.js +60 -0
- package/dist/sandbox-3AG2WJLM.js +94 -0
- package/dist/sandbox-5JOFBS75.js +81 -0
- package/dist/schema-extensions.sql +234 -0
- package/dist/security-2KI5XNDG.js +17 -0
- package/dist/setup-bb-plugin-QAQWZCV5.js +738 -0
- package/dist/setup-claude-code-P7BTTSPO.js +317 -0
- package/dist/setup-temp-cleanup-GASCWZ2I.js +20 -0
- package/dist/shared-utils-D6U6UREP.js +37 -0
- package/dist/simple-artifact-goal-OTXNGZY6.js +25 -0
- package/dist/sprint-QGXLN3SO.js +338 -0
- package/dist/sprint-bridge-FNKQCJGL.js +68 -0
- package/dist/sprint-execution-service-4ELKNJDS.js +311 -0
- package/dist/sprint-manager-7TNCXCIX.js +91 -0
- package/dist/sqlite-wrapper-UDWCTNLJ.js +17 -0
- package/dist/stale-cleanup-orchestrator-H5XKGZP2.js +62 -0
- package/dist/storage-access-F6ALDN4S.js +22 -0
- package/dist/synthesis-TGQUVAW3.js +185 -0
- package/dist/task-classifier-DZIRBSS2.js +494 -0
- package/dist/task-list-fast-PJ2EXVZE.js +147 -0
- package/dist/task-manager-5742GMJK.js +93 -0
- package/dist/task-splitter-6MBXSWF6.js +79 -0
- package/dist/task-type-detector-OLVOMXZO.js +21 -0
- package/dist/task-type-resolver-4HG6TXFV.js +42 -0
- package/dist/telemetry-JX7EMNOZ.js +131 -0
- package/dist/telemetry-health-tracker-DMSSGBZ2.js +11 -0
- package/dist/telemetry-path-resolver-7ICSG276.js +28 -0
- package/dist/telemetry-reliability-2MLJ4J6B.js +125 -0
- package/dist/token-estimator-PDU345YE.js +37 -0
- package/dist/transports-3UEO6C5Y.js +255 -0
- package/dist/unified-cost-tracker-F3W5NMPG.js +35 -0
- package/dist/unified-state-cleanup-LJPF2BOW.js +276 -0
- package/dist/universal-cost-calculator-BAJV35GA.js +42 -0
- package/dist/user-agents-GHUGCSKN.js +22 -0
- package/dist/variant-delivery-state-XCOYVCM4.js +33 -0
- package/dist/variant-state-cleanup-QIEKRULB.js +280 -0
- package/dist/variant-success-PJXWE3ZV.js +12 -0
- package/dist/variant-worker-HPTUA37Z.js +4196 -0
- package/dist/variant-worker.js +24 -0
- package/dist/variants-2HDYZOCG.js +233 -0
- package/dist/verification-toolchains-VVKT3GBF.js +28 -0
- package/dist/verify-commands-5F6CJV4B.js +37 -0
- package/dist/warning-EGQGUUIO.js +12 -0
- package/dist/work-edge-RWEDWJZF.js +49 -0
- package/dist/work-item-writer-WHAVZXAY.js +19 -0
- package/dist/worker-AGWHD4D2.js +1435 -0
- package/dist/worker.js +19 -0
- package/dist/worktree-metrics-3DFUYVYE.js +26 -0
- package/dist/writeback-VZKHYHSN.js +44 -0
- package/docs/CLI_REFERENCE.md +7040 -0
- package/docs/PROVIDER_SETUP.md +1570 -0
- package/docs/README.md +96 -0
- package/docs/TROUBLESHOOTING.md +2232 -0
- package/docs/getting-started/QUICK_START.md +294 -0
- package/docs/getting-started/README.md +188 -0
- package/docs/getting-started/SETUP.md +52 -0
- package/docs/reference/AGENT_CONTEXT.md +38 -0
- package/docs/reference/PROVIDER_RELEASE_TIERS.md +34 -0
- package/docs/reference/README.md +381 -0
- package/docs/reference/SECURITY.md +262 -0
- package/integrations/bb-plugin-poetic/README.md +362 -0
- package/integrations/bb-plugin-poetic/app.css +341 -0
- package/integrations/bb-plugin-poetic/app.tsx +5059 -0
- package/integrations/bb-plugin-poetic/package.json +45 -0
- package/integrations/bb-plugin-poetic/server.ts +444 -0
- package/integrations/bb-plugin-poetic/src/adapter.ts +3312 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-model.ts +150 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-schema.ts +355 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-service.ts +419 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-view.ts +1026 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-schema.ts +122 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-service.ts +105 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-view.ts +138 -0
- package/integrations/bb-plugin-poetic/src/contract.ts +615 -0
- package/integrations/bb-plugin-poetic/src/finalize-model.ts +205 -0
- package/integrations/bb-plugin-poetic/src/finalize-schema.ts +158 -0
- package/integrations/bb-plugin-poetic/src/finalize-service.ts +382 -0
- package/integrations/bb-plugin-poetic/src/finalize-view.ts +159 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-readback.ts +303 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-schema.ts +33 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-view.ts +459 -0
- package/integrations/bb-plugin-poetic/src/model.ts +1302 -0
- package/integrations/bb-plugin-poetic/src/monitor-service.ts +233 -0
- package/integrations/bb-plugin-poetic/src/panel-read-ux.ts +97 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-schema.ts +166 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-service.ts +480 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-view.ts +109 -0
- package/integrations/bb-plugin-poetic/src/planning-readback.ts +204 -0
- package/integrations/bb-plugin-poetic/src/planning-service.ts +594 -0
- package/integrations/bb-plugin-poetic/src/planning-workspace-schema.ts +95 -0
- package/integrations/bb-plugin-poetic/src/planning-workspace.ts +323 -0
- package/integrations/bb-plugin-poetic/src/poll-handoff.ts +136 -0
- package/integrations/bb-plugin-poetic/src/project-target-server.ts +21 -0
- package/integrations/bb-plugin-poetic/src/project-target.ts +45 -0
- package/integrations/bb-plugin-poetic/src/reference-index.ts +192 -0
- package/integrations/bb-plugin-poetic/src/repository-schema.ts +30 -0
- package/integrations/bb-plugin-poetic/src/result-explorer-view.ts +597 -0
- package/integrations/bb-plugin-poetic/src/result-readback-model.ts +163 -0
- package/integrations/bb-plugin-poetic/src/result-readback-schema.ts +319 -0
- package/integrations/bb-plugin-poetic/src/result-readback-service.ts +333 -0
- package/integrations/bb-plugin-poetic/src/run-page-view.ts +177 -0
- package/integrations/bb-plugin-poetic/src/setup-config-schema.ts +291 -0
- package/integrations/bb-plugin-poetic/src/setup-config-service.ts +590 -0
- package/integrations/bb-plugin-poetic/src/setup-config-view.ts +413 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-model.ts +150 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-schema.ts +136 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-service.ts +341 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-view.ts +116 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-model.ts +714 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-readback.ts +231 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-schema.ts +601 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-service.ts +775 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-view.ts +1221 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-model.ts +567 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-schema.ts +635 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-service.ts +842 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-view.ts +1227 -0
- package/integrations/bb-plugin-poetic/src/task-detail-readback.ts +56 -0
- package/integrations/bb-plugin-poetic/src/task-detail-schema.ts +144 -0
- package/integrations/bb-plugin-poetic/src/task-navigation.ts +446 -0
- package/integrations/bb-plugin-poetic/src/workbench-route.ts +74 -0
- package/integrations/bb-plugin-poetic/tests/host-contract.test.ts +773 -0
- package/integrations/bb-plugin-poetic/tsconfig.json +18 -0
- package/integrations/bb-plugin-poetic/types/PROVENANCE.json +52 -0
- package/integrations/bb-plugin-poetic/types/bb-plugin-sdk-app.d.ts +1444 -0
- package/integrations/bb-plugin-poetic/types/bb-plugin-sdk.d.ts +13030 -0
- package/integrations/bb-plugin-poetic/vitest.host.config.ts +71 -0
- package/package.json +331 -0
- package/schemas/README.md +80 -0
- package/schemas/config-v1.schema.json +136 -0
- package/schemas/execution-config.schema.json +47 -0
- package/schemas/judge-scoring.schema.json +370 -0
- package/schemas/poetic.config.schema.json +2214 -0
- package/schemas/provider-config.schema.json +203 -0
- package/schemas/telemetry-config.schema.json +79 -0
- package/schemas/user-preferences.schema.json +141 -0
- package/scripts/assert-node-runtime.mjs +140 -0
- package/scripts/preflight-native.mjs +49 -0
- package/scripts/preinstall-node-check.mjs +78 -0
- package/scripts/setup-git-hooks.mjs +24 -0
- package/scripts/sync.sh +2722 -0
- package/scripts/write-node-launcher.sh +108 -0
- package/src/resources/gemini/slash-packs/default/plan.toml +15 -0
- package/src/resources/gemini/slash-packs/default/summary.toml +16 -0
- package/src/resources/gemini/slash-packs/default/tests.toml +16 -0
- package/templates/.poetic/README.md +37 -0
- package/templates/.poetic/agents/README.md +296 -0
- package/templates/.poetic/agents/api-documenter.md +147 -0
- package/templates/.poetic/agents/backend-architect.md +31 -0
- package/templates/.poetic/agents/code-reviewer.md +157 -0
- package/templates/.poetic/agents/data-scientist.md +179 -0
- package/templates/.poetic/agents/database-optimizer.md +145 -0
- package/templates/.poetic/agents/debugger.md +31 -0
- package/templates/.poetic/agents/deployment-engineer.md +164 -0
- package/templates/.poetic/agents/devops-troubleshooter.md +139 -0
- package/templates/.poetic/agents/frontend-developer.md +150 -0
- package/templates/.poetic/agents/javascript-pro.md +36 -0
- package/templates/.poetic/agents/performance-engineer.md +151 -0
- package/templates/.poetic/agents/python-pro.md +137 -0
- package/templates/.poetic/agents/test-automator.md +147 -0
- package/templates/.poetic/agents/typescript-pro.md +34 -0
- package/templates/.poetic/config/poetic.config.jsonc +69 -0
- package/templates/.poetic/config/project-context.template.json +6 -0
- package/templates/.poetic/config/task-type-aliases.presets/kanban.yaml +14 -0
- package/templates/.poetic/config/task-type-aliases.presets/scrum.yaml +17 -0
- package/templates/.poetic/config/task-type-aliases.presets/xp.yaml +12 -0
- package/templates/.poetic/config/task-type-aliases.yaml +28 -0
- package/templates/.poetic/gitignore.template +55 -0
- package/templates/.poetic/task-types/analysis.yaml +40 -0
- package/templates/.poetic/task-types/architecture.yaml +38 -0
- package/templates/.poetic/task-types/doc.yaml +35 -0
- package/templates/.poetic/task-types/feature.yaml +23 -0
- package/templates/.poetic/task-types/general.yaml +6 -0
- package/templates/.poetic/task-types/security.yaml +39 -0
- package/templates/AGENTS.template.md +99 -0
- package/templates/CLAUDE.template.md +1 -0
- package/templates/GEMINI.template.md +1 -0
- package/templates/builtin-workflows/code-review.yaml +73 -0
- package/templates/builtin-workflows/compete-streak.yaml +78 -0
- package/templates/builtin-workflows/hello-verify.yaml +10 -0
- package/templates/builtin-workflows/judge-regression.yaml +114 -0
- package/templates/builtin-workflows/skills/code-review/SKILL.md +60 -0
- package/templates/builtin-workflows/skills/hello-verify/SKILL.md +6 -0
- package/templates/guard-kit/GUARD_SETUP.md.template +255 -0
- package/templates/guard-kit/check.mjs.template +777 -0
- package/templates/guard-kit/config.json.template +6 -0
- package/templates/guard-kit/poetic-guard.yml.template +189 -0
- package/templates/profiles/README.md +56 -0
- package/templates/profiles/frontier-claude.json +27 -0
- package/templates/profiles/frontier-codex.json +26 -0
|
@@ -0,0 +1,1905 @@
|
|
|
1
|
+
import {
|
|
2
|
+
anonymizeVariants,
|
|
3
|
+
buildDeterministicPreScoreGateSection,
|
|
4
|
+
buildVariantDeliverableExcerptForJudge,
|
|
5
|
+
buildVariantEvidenceCandidatesForJudge,
|
|
6
|
+
checkRankingScoreConsistency,
|
|
7
|
+
computeDeterministicPreScoreGateContext,
|
|
8
|
+
computePromptOrderEvidenceCandidateLimit,
|
|
9
|
+
computeSelfConsistencyWarning,
|
|
10
|
+
createFallbackAnonymizationSeed,
|
|
11
|
+
createPercentScore,
|
|
12
|
+
deAnonymizeRationale,
|
|
13
|
+
detectDuplicateContent,
|
|
14
|
+
getDefaultJudgeModel,
|
|
15
|
+
normalizeRankingIds,
|
|
16
|
+
observeVariantHealth,
|
|
17
|
+
persistJudgeFailureArtifact,
|
|
18
|
+
resolveAnonymizationRunId,
|
|
19
|
+
selectDeterministicWinnerFromScores,
|
|
20
|
+
shouldAnonymize,
|
|
21
|
+
validateRanking
|
|
22
|
+
} from "./chunk-TSPQONRT.js";
|
|
23
|
+
import {
|
|
24
|
+
charsToTokens
|
|
25
|
+
} from "./chunk-LFTQVYBB.js";
|
|
26
|
+
import {
|
|
27
|
+
THREE_BUCKET_RUBRIC_FORMULA_TEXT,
|
|
28
|
+
computeFinalScore
|
|
29
|
+
} from "./chunk-73NCTWKL.js";
|
|
30
|
+
import {
|
|
31
|
+
collectJsonCandidates,
|
|
32
|
+
isStandaloneMalformedJsonCandidate,
|
|
33
|
+
parseJsonCandidate
|
|
34
|
+
} from "./chunk-VVGEPBPS.js";
|
|
35
|
+
import {
|
|
36
|
+
buildVariantFileInventory
|
|
37
|
+
} from "./chunk-E74LUIID.js";
|
|
38
|
+
import {
|
|
39
|
+
formatVerificationProvenanceForPrompt,
|
|
40
|
+
resolveJudgeEvaluationContext
|
|
41
|
+
} from "./chunk-66AGFTBQ.js";
|
|
42
|
+
import {
|
|
43
|
+
extractChangedFilesMeta
|
|
44
|
+
} from "./chunk-CFBIG37O.js";
|
|
45
|
+
import {
|
|
46
|
+
resolveTaskType
|
|
47
|
+
} from "./chunk-W4OMESPG.js";
|
|
48
|
+
import {
|
|
49
|
+
detectExplicitReviewTask,
|
|
50
|
+
parsePromptSpec
|
|
51
|
+
} from "./chunk-X6UIEADS.js";
|
|
52
|
+
import {
|
|
53
|
+
LlmJudgeExecutor
|
|
54
|
+
} from "./chunk-5REDMNLZ.js";
|
|
55
|
+
import {
|
|
56
|
+
mergeTokenUsage,
|
|
57
|
+
sumTokenUsage
|
|
58
|
+
} from "./chunk-7DNSNKJV.js";
|
|
59
|
+
import {
|
|
60
|
+
extractDiffText,
|
|
61
|
+
frameUntrustedJudgeTaskText
|
|
62
|
+
} from "./chunk-IATQ7NBI.js";
|
|
63
|
+
import {
|
|
64
|
+
JUDGE_FEATURE_FLAGS
|
|
65
|
+
} from "./chunk-BSE4R6XD.js";
|
|
66
|
+
|
|
67
|
+
// src/core/judge/ai-judge-prompt.ts
|
|
68
|
+
function buildDefaultPromptText() {
|
|
69
|
+
const emptyArrayInstruction = JUDGE_FEATURE_FLAGS.RATIONALE_EMPTY_ARRAYS === "on" ? "\n15. If a variant has no genuine strengths, weaknesses, or evidence for a dimension, use an empty array [] rather than inventing items." : "";
|
|
70
|
+
return `You are an impartial judge evaluating submissions for a task.
|
|
71
|
+
The task may involve software, design, writing, research, planning, or other knowledge work.
|
|
72
|
+
|
|
73
|
+
## Task
|
|
74
|
+
{task_prompt}
|
|
75
|
+
|
|
76
|
+
## Task Profile Guidance
|
|
77
|
+
{task_profile_guidance}
|
|
78
|
+
|
|
79
|
+
## Submissions
|
|
80
|
+
{variant_sections}
|
|
81
|
+
|
|
82
|
+
## Evaluation Criteria
|
|
83
|
+
|
|
84
|
+
Rate each submission on three dimensions:
|
|
85
|
+
|
|
86
|
+
**Delivery (50%)**: Did the submission produce the requested deliverable?
|
|
87
|
+
- 90-100: All requirements met, task fully completed
|
|
88
|
+
- 70-89: Most requirements met, minor gaps
|
|
89
|
+
- 50-69: Partial completion, meaningful gaps
|
|
90
|
+
- 30-49: Minimal delivery, major gaps
|
|
91
|
+
- 0-29: Failed or no meaningful output
|
|
92
|
+
If the task requests implementation and a submission only describes an approach without producing the deliverable, treat that as important delivery evidence; do not apply a fixed host-side cap. For tasks classified as analysis/planning/design in the Task Profile Guidance, a well-structured analysis IS the deliverable.
|
|
93
|
+
If the task explicitly calls for a minimal or tiny change (e.g., a single comment/doc/typo fix), unnecessary extra scope (new files/APIs or unrelated changes) should reduce the Delivery score.
|
|
94
|
+
When the task requires listing the exact resources that would be affected (branch names, paths, PIDs, file lists, identifiers), placeholder or generated identifiers do not satisfy the requirement. Examples of fabrication: identifiers built from a template expression combining a fixed prefix with a loop index, synthetic indices generated by Array.from or similar, or literal placeholder strings (such as fake paths or fake resource names). Treat fabricated identifiers as a delivery defect, not a cosmetic one.
|
|
95
|
+
|
|
96
|
+
**Correctness (30%)**: Is the work likely correct?
|
|
97
|
+
- 90-100: High confidence in correctness, edge cases addressed
|
|
98
|
+
- 70-89: Good correctness, minor concerns
|
|
99
|
+
- 50-69: Some correctness risks
|
|
100
|
+
- 30-49: Significant risks or errors
|
|
101
|
+
- 0-29: Fundamental errors
|
|
102
|
+
For CLI, configuration, or API behavior tasks: named flags, options, or parameters must be traced from declaration through parsing into the handler that is supposed to read them. A flag that is declared but never read by the code path it is supposed to gate is a correctness defect, not a documentation issue, even when the variant's tests pass against mocks.
|
|
103
|
+
|
|
104
|
+
**Quality (20%)**: Is the work clear, well-structured, and appropriate for its domain?
|
|
105
|
+
- 90-100: Excellent structure and clarity
|
|
106
|
+
- 70-89: Good quality, minor issues
|
|
107
|
+
- 50-69: Acceptable, some concerns
|
|
108
|
+
- 30-49: Notable quality problems
|
|
109
|
+
- 0-29: Poor quality
|
|
110
|
+
|
|
111
|
+
## Evidence Calibration
|
|
112
|
+
|
|
113
|
+
Distinguish between proxy evidence and outcome evidence.
|
|
114
|
+
- Proxy evidence: naming alignment, documentation updates, keyword matches, structural resemblance to the requested change.
|
|
115
|
+
- Outcome evidence: direct proof that the requested result was achieved \u2014 behavior changed, failing condition removed, verification passed.
|
|
116
|
+
|
|
117
|
+
Outcome evidence is stronger than proxy evidence. Multiple corroborating proxies can still be jointly wrong when none directly measures the requested outcome. Do not treat documentation quality or requirement-word overlap as proof that the task outcome was achieved. When two variants achieve the same outcome, proxy evidence may still differentiate quality.
|
|
118
|
+
When the task explicitly names representative commands, operations, or target flows, structural changes alone are still only proxy evidence. If direct evidence for those named targets is missing, contradictory, or only implied, state that gap explicitly in the rationale and keep confidence below strong-certainty levels.
|
|
119
|
+
Mocked tests are proxy evidence, not outcome evidence. A test that mocks the module under inspection is proxy evidence for that module's runtime behavior, regardless of assertion count or quality. Outcome evidence requires real execution signals (gate results, logs, stdout), verified integration behavior, or direct unmocked code-path proof. Do not upgrade proxy to outcome because of test count, naming overlap, or structural match.
|
|
120
|
+
When task-appropriate deterministic verification passes and the evidence shows the requested requirements implemented, do not invent a weakness from excerpt visibility alone (for example, "not explicitly visible in diff excerpt"). Penalize only concrete task-relevant defects, unmet requirements, contradictory evidence, or real verification gaps.
|
|
121
|
+
|
|
122
|
+
## Instructions
|
|
123
|
+
|
|
124
|
+
1. For each submission, identify specific evidence from the deliverable excerpt.
|
|
125
|
+
Cite evidence using E1, E2, etc. from the evidence candidates.
|
|
126
|
+
2. Assess each criterion against the evidence you identified.
|
|
127
|
+
3. Derive scores from your assessment. The score should follow from the evidence, not precede it.
|
|
128
|
+
4. Judge by the task's actual domain. For non-technical tasks, do not apply software-centric assumptions.
|
|
129
|
+
5. Submission order is arbitrary -- do not let presentation order influence scores.
|
|
130
|
+
6. Length is not quality. Score substance, not volume. Keep each criterion rationale to 1-2 sentences. Evidence arrays: cite at most 3 IDs per criterion. Summary rationale and rankingRationale: 3 sentences maximum; cite at least one evidence ID (E1, E2...) or file path when available. Brevity prevents response truncation.
|
|
131
|
+
7. If evaluating a single submission, score on absolute merit. Do not inflate scores due to lack of competition. Similarly, do not inflate confidence: for single-variant evaluations, confidence reflects quality assessment certainty (how certain you are the scores are accurate), NOT winner selection certainty (which is trivially 1.0 and must not be reported as the confidence value). When the sole submission produced no output (empty diff) or all dimension scores are near zero (\u2264 10), confidence MUST be \u2264 0.40 \u2014 there is insufficient evidence to assess quality meaningfully.
|
|
132
|
+
8. Provide confidence (0.0-1.0) in the score ranking:
|
|
133
|
+
- 0.95+: Unambiguous winner, nearly any evaluator would agree
|
|
134
|
+
- 0.85-0.94: Strong direct evidence supports the winner (e.g., deliverables produced vs none, on-task vs off-task, core feature functional vs non-functional, required targets directly covered); minor residual uncertainty
|
|
135
|
+
- 0.70-0.84: Clear lean toward winner; evidence supports the ranking but some dimensions are close
|
|
136
|
+
- 0.50-0.69: Meaningful uncertainty; winner is plausible but alternative ranking defensible
|
|
137
|
+
- 0.30-0.49: Close call or insufficient evidence
|
|
138
|
+
- Below 0.30: Single-variant with no output or zero scores \u2014 quality cannot be assessed
|
|
139
|
+
When direct outcome evidence is decisive (deliverables produced vs none, on-task vs off-task, passing tests vs failing, core feature functional vs non-functional), confidence should be at least 0.85 even if individual evidence excerpts are thin. Do not apply that high-confidence floor when the task names specific commands or operations and the evidence remains only structural or proxy-level for those targets.
|
|
140
|
+
9. Score calibration:
|
|
141
|
+
- Reserve 100 for cases with no material weaknesses found from evidence.
|
|
142
|
+
- Use 95+ only when evidence quality is high and no unresolved must-level gaps are visible.
|
|
143
|
+
10. For each variant, include one concrete limitation/trade-off with evidence.
|
|
144
|
+
If none are found, explicitly state "No material weakness found from available evidence."
|
|
145
|
+
11. Confidence rationale must reference evidence quality and residual uncertainty.
|
|
146
|
+
12. Include a top-2 separation rationale explaining why the top-scoring variant beat runner-up.
|
|
147
|
+
For single-variant evaluation, set top-2 rationale to "N/A: single variant".
|
|
148
|
+
13. Declare the winner in the top-level winner object when comparing multiple variants. If no variant should win, return \`"winner": null\` and explain why in \`summary.rankingRationale\`. An explicit null is authoritative; summary.ranking remains ordered comparative context and is the fallback only when winner is omitted or invalid.
|
|
149
|
+
14. Rank all variants from best to worst in summary.ranking.
|
|
150
|
+
15. For multi-variant evaluations, compute os (overallScore) per variant as: ${THREE_BUCKET_RUBRIC_FORMULA_TEXT}. summary.ranking MUST normally be sorted by os descending. When a variant has a decisive correctness failure (e.g., failing tests, broken build, fundamental logic errors, inaccurate findings with line-citation discrepancies), reflect that failure clearly in the correctness score so that the os ranking captures the quality difference. Explain in rankingRationale when a correctness failure is the deciding factor.${emptyArrayInstruction}
|
|
151
|
+
16. Include the optional dimensions block for every variant. Use independent 0-100 scores for readability, maintainability, security, and innovation. Do not copy the same number into every field unless the evidence genuinely supports identical scores.
|
|
152
|
+
|
|
153
|
+
Respond with ONLY the following JSON structure. First character must be '{'. No prose, no markdown fences, no explanatory text.
|
|
154
|
+
All "score" fields, the "dimensions" values, and "os" are integers 0-100; "confidence" is between 0.0 and 1.0. The example values below are illustrative \u2014 replace them with your real assessment:
|
|
155
|
+
{
|
|
156
|
+
"variants": [
|
|
157
|
+
{
|
|
158
|
+
"id": "variant-id",
|
|
159
|
+
"delivery": { "score": 82, "rationale": "...", "evidence": ["E1"] },
|
|
160
|
+
"correctness": { "score": 76, "rationale": "...", "evidence": ["E2"] },
|
|
161
|
+
"quality": { "score": 71, "rationale": "...", "evidence": [] },
|
|
162
|
+
"dimensions": {
|
|
163
|
+
"readability": 74,
|
|
164
|
+
"maintainability": 70,
|
|
165
|
+
"security": 66,
|
|
166
|
+
"innovation": 62
|
|
167
|
+
},
|
|
168
|
+
"limitation": { "summary": "...", "evidence": ["E3"] },
|
|
169
|
+
"os": 79
|
|
170
|
+
}
|
|
171
|
+
],
|
|
172
|
+
"summary": {
|
|
173
|
+
"ranking": ["variant-best", "variant-worst"],
|
|
174
|
+
"rankingRationale": "...",
|
|
175
|
+
"confidence": 0.85,
|
|
176
|
+
"confidenceRationale": "...",
|
|
177
|
+
"top2SeparationRationale": "..."
|
|
178
|
+
},
|
|
179
|
+
"winner": {
|
|
180
|
+
"id": "variant-best",
|
|
181
|
+
"reason": "..."
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
For a judge-authored no-winner decision, replace the winner object with \`"winner": null\`; still include every variant in summary.ranking.`;
|
|
185
|
+
}
|
|
186
|
+
var DEFAULT_AI_JUDGE_PROMPT = buildDefaultPromptText();
|
|
187
|
+
function buildTaskProfileGuidance(context) {
|
|
188
|
+
const base = `Task profile: ${context.taskType} (detected confidence: ${context.confidence}). This profile is prompt-only context and must not be treated as a deterministic scoring rule.`;
|
|
189
|
+
if (context.taskType === "code") {
|
|
190
|
+
return `${base}
|
|
191
|
+
- Prioritize direct evidence of requirements coverage and likely behavior.
|
|
192
|
+
- Use execution-related evidence when present.
|
|
193
|
+
- Do not infer lower quality from missing execution evidence unless execution was explicitly requested.`;
|
|
194
|
+
}
|
|
195
|
+
if (context.limitedEmpiricalSignals) {
|
|
196
|
+
return `${base}
|
|
197
|
+
Limited empirical signals are available for this task type.
|
|
198
|
+
- Weight deliverable content and explicit requirements coverage heavily.
|
|
199
|
+
- Do not infer lower quality from missing execution-specific signals.
|
|
200
|
+
- Prefer concrete, checkable claims over speculative assertions.`;
|
|
201
|
+
}
|
|
202
|
+
return `${base}
|
|
203
|
+
- Weight deliverable content and explicit requirements coverage heavily.
|
|
204
|
+
- Prefer concrete, checkable claims over speculative assertions.`;
|
|
205
|
+
}
|
|
206
|
+
function buildDomainSignalSemanticsBlock(context) {
|
|
207
|
+
const combined = `${context.taskPrompt}
|
|
208
|
+
${context.variantSections}`;
|
|
209
|
+
if (!/\b(?:synthesisApplied|winnerFallback|aiGuidedFallbackAttempted|aiGuidedFallbackOutcome|actualStrategy|aiFirstActualStrategy|donor_augmented|winner_equivalent|candidate_rejected|generation_failed)\b/.test(
|
|
210
|
+
combined
|
|
211
|
+
)) {
|
|
212
|
+
return null;
|
|
213
|
+
}
|
|
214
|
+
return `## Relevant Existing Signal Semantics
|
|
215
|
+
|
|
216
|
+
The task/submissions reference existing synthesis visibility signals. Treat these as factual context for interpretation:
|
|
217
|
+
- \`synthesisApplied=true\` means a synthesized result was applied; it does not by itself prove donor material survived.
|
|
218
|
+
- \`winnerFallback=true\` means the declared winner was retained.
|
|
219
|
+
- \`aiGuidedFallbackOutcome=donor_augmented\` means agentic inspection retained donor material.
|
|
220
|
+
- \`aiGuidedFallbackOutcome=winner_equivalent\` means agentic inspection found no useful donor delta and retained the winner.
|
|
221
|
+
- \`aiGuidedFallbackOutcome=generation_failed\` or \`candidate_rejected\` means fallback followed generation, apply, or gate failure.
|
|
222
|
+
- \`actualStrategy\` / \`aiFirstActualStrategy\` describe the actual path used by agentic synthesis: agentic-synthesis, cherry-pick, or winner-only.
|
|
223
|
+
|
|
224
|
+
Use these semantics only when judging claims about these fields. They are contextual evidence, not scoring caps.`;
|
|
225
|
+
}
|
|
226
|
+
function buildAIJudgePrompt(context, customPrompt) {
|
|
227
|
+
const template = customPrompt ?? DEFAULT_AI_JUDGE_PROMPT;
|
|
228
|
+
const safeTaskPrompt = frameUntrustedJudgeTaskText(context.taskPrompt ?? "");
|
|
229
|
+
const guidance = context.taskProfileGuidance?.trim() || "No additional task profile guidance.";
|
|
230
|
+
const templatePlaceholderProbe = template.replace(/\{task_prompt\}/g, "__TASK_PROMPT__").replace(/\{variant_sections\}/g, "__VARIANT_SECTIONS__").replace(/\{task_profile_guidance\}/g, "__TASK_PROFILE_GUIDANCE__").replace(/\{evaluation_guidance\}/g, "__TASK_PROFILE_GUIDANCE__");
|
|
231
|
+
const unreplacedTemplateTokens = templatePlaceholderProbe.match(/\{[a-z_]+\}/g);
|
|
232
|
+
if (unreplacedTemplateTokens) {
|
|
233
|
+
throw new Error(
|
|
234
|
+
`AIJudgePrompt: unreplaced placeholder tokens in prompt: ${unreplacedTemplateTokens.join(", ")}`
|
|
235
|
+
);
|
|
236
|
+
}
|
|
237
|
+
const hadPlaceholder = template.includes("{task_profile_guidance}") || template.includes("{evaluation_guidance}");
|
|
238
|
+
let result = template.replace(/\{task_prompt\}/g, safeTaskPrompt).replace(/\{variant_sections\}/g, context.variantSections).replace(/\{task_profile_guidance\}/g, guidance).replace(/\{evaluation_guidance\}/g, guidance);
|
|
239
|
+
if (!hadPlaceholder && context.taskProfileGuidance) {
|
|
240
|
+
result = `${result}
|
|
241
|
+
|
|
242
|
+
## Task Profile Guidance
|
|
243
|
+
${guidance}`;
|
|
244
|
+
}
|
|
245
|
+
if (context.judgeProfile) {
|
|
246
|
+
const profileBlock = buildJudgeProfileBlock(context.judgeProfile);
|
|
247
|
+
if (profileBlock) {
|
|
248
|
+
result = `${result}
|
|
249
|
+
|
|
250
|
+
${profileBlock}`;
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
const semanticsBlock = buildDomainSignalSemanticsBlock(context);
|
|
254
|
+
if (semanticsBlock) {
|
|
255
|
+
result = `${result}
|
|
256
|
+
|
|
257
|
+
${semanticsBlock}`;
|
|
258
|
+
}
|
|
259
|
+
return result;
|
|
260
|
+
}
|
|
261
|
+
function buildJudgeProfileBlock(profile) {
|
|
262
|
+
if (profile === "best-as-delivered") {
|
|
263
|
+
return `## Judge Profile: BEST AS DELIVERED
|
|
264
|
+
|
|
265
|
+
Declare the best complete deliverable as submitted.
|
|
266
|
+
Evaluate what was ACTUALLY delivered, not what COULD be delivered with follow-up work.
|
|
267
|
+
Penalize unrelated artifact churn, unmanaged/generated/state artifacts, missing integration points,
|
|
268
|
+
incomplete requested verification or presentation wiring, and cleanup required before accepting.
|
|
269
|
+
Do not give credit for unrealized potential that requires additional work to realize.`;
|
|
270
|
+
}
|
|
271
|
+
if (profile === "acceptance-ready") {
|
|
272
|
+
return `## Judge Profile: ACCEPTANCE READY
|
|
273
|
+
|
|
274
|
+
Declare the winner that can be accepted now with the least unresolved risk.
|
|
275
|
+
Treat passing task-appropriate verification as strong evidence of acceptance readiness.
|
|
276
|
+
Treat failing or missing required verification signals as major acceptance risk.
|
|
277
|
+
Do not select a failing variant over a passing variant unless all passing variants are materially incomplete or unsafe, and explain why.`;
|
|
278
|
+
}
|
|
279
|
+
if (profile === "best-foundation") {
|
|
280
|
+
return `## Judge Profile: BEST FOUNDATION
|
|
281
|
+
|
|
282
|
+
Declare the winner that is the best base for a follow-up repair or synthesis pass.
|
|
283
|
+
Do not ignore failing verification \u2014 classify the apparent follow-up cost (mechanical, localized, moderate, structural).
|
|
284
|
+
A failing variant may beat a passing variant only when its underlying approach, requirement coverage, and locality make total acceptance effort lower than starting from the passing candidate.`;
|
|
285
|
+
}
|
|
286
|
+
return "";
|
|
287
|
+
}
|
|
288
|
+
function formatVariantSection(variantId, variantName, evidenceList, deliverableExcerpt, deterministicPreScoreGates) {
|
|
289
|
+
const lines = [];
|
|
290
|
+
lines.push(`### Variant: ${variantName} (id: ${variantId})`);
|
|
291
|
+
lines.push("");
|
|
292
|
+
if (typeof deterministicPreScoreGates === "string" && deterministicPreScoreGates.trim().length > 0) {
|
|
293
|
+
lines.push("**Deterministic Pre-Score Gates**:");
|
|
294
|
+
lines.push(deterministicPreScoreGates.trim());
|
|
295
|
+
lines.push("");
|
|
296
|
+
}
|
|
297
|
+
const promptOrderMode = JUDGE_FEATURE_FLAGS.JUDGE_PROFILE;
|
|
298
|
+
if (promptOrderMode === "stable" || promptOrderMode === "shadow" || promptOrderMode === "experimental") {
|
|
299
|
+
lines.push("**Deliverable Excerpt**:");
|
|
300
|
+
lines.push(deliverableExcerpt);
|
|
301
|
+
lines.push("");
|
|
302
|
+
if (evidenceList.length > 0) {
|
|
303
|
+
lines.push("**Evidence Candidates** (cite as E1, E2, etc.):");
|
|
304
|
+
evidenceList.forEach((evidence, index) => {
|
|
305
|
+
lines.push(`E${index + 1}: ${evidence}`);
|
|
306
|
+
});
|
|
307
|
+
lines.push("");
|
|
308
|
+
}
|
|
309
|
+
} else {
|
|
310
|
+
if (evidenceList.length > 0) {
|
|
311
|
+
lines.push("**Evidence Candidates** (cite as E1, E2, etc.):");
|
|
312
|
+
evidenceList.forEach((evidence, index) => {
|
|
313
|
+
lines.push(`E${index + 1}: ${evidence}`);
|
|
314
|
+
});
|
|
315
|
+
lines.push("");
|
|
316
|
+
}
|
|
317
|
+
lines.push("**Deliverable Excerpt**:");
|
|
318
|
+
lines.push(deliverableExcerpt);
|
|
319
|
+
lines.push("");
|
|
320
|
+
}
|
|
321
|
+
return lines.join("\n");
|
|
322
|
+
}
|
|
323
|
+
function validateAIJudgeOutput(output) {
|
|
324
|
+
if (!output || typeof output !== "object") {
|
|
325
|
+
return false;
|
|
326
|
+
}
|
|
327
|
+
const root = output;
|
|
328
|
+
if (!Array.isArray(root.variants) || root.variants.length === 0) {
|
|
329
|
+
return false;
|
|
330
|
+
}
|
|
331
|
+
for (const variant of root.variants) {
|
|
332
|
+
if (!validateVariantEntry(variant)) {
|
|
333
|
+
return false;
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
if (root.variants.length > 1) {
|
|
337
|
+
for (const variant of root.variants) {
|
|
338
|
+
const v = variant;
|
|
339
|
+
if (v.os === void 0) return false;
|
|
340
|
+
}
|
|
341
|
+
const summary = root.summary;
|
|
342
|
+
if (!summary || !Array.isArray(summary?.ranking) || summary.ranking.length === 0) {
|
|
343
|
+
return false;
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
if (!validateSummaryEntry(root.summary)) {
|
|
347
|
+
return false;
|
|
348
|
+
}
|
|
349
|
+
if (root.ranking !== void 0 && !validateRankingArray(root.ranking)) {
|
|
350
|
+
return false;
|
|
351
|
+
}
|
|
352
|
+
if (!validateWinnerEntry(root.winner)) {
|
|
353
|
+
return false;
|
|
354
|
+
}
|
|
355
|
+
return true;
|
|
356
|
+
}
|
|
357
|
+
function validateVariantEntry(variant) {
|
|
358
|
+
if (!variant || typeof variant !== "object") {
|
|
359
|
+
return false;
|
|
360
|
+
}
|
|
361
|
+
const v = variant;
|
|
362
|
+
if (typeof v.id !== "string" || v.id.length === 0) {
|
|
363
|
+
return false;
|
|
364
|
+
}
|
|
365
|
+
if (!validateBucketEntry(v.delivery)) {
|
|
366
|
+
return false;
|
|
367
|
+
}
|
|
368
|
+
if (!validateBucketEntry(v.correctness)) {
|
|
369
|
+
return false;
|
|
370
|
+
}
|
|
371
|
+
if (!validateBucketEntry(v.quality)) {
|
|
372
|
+
return false;
|
|
373
|
+
}
|
|
374
|
+
if (!validateLimitationEntry(v.limitation)) {
|
|
375
|
+
return false;
|
|
376
|
+
}
|
|
377
|
+
if (!validateDimensionsEntry(v.dimensions)) {
|
|
378
|
+
return false;
|
|
379
|
+
}
|
|
380
|
+
if (v.os !== void 0) {
|
|
381
|
+
if (typeof v.os !== "number" || !Number.isFinite(v.os) || v.os < 0 || v.os > 100) {
|
|
382
|
+
return false;
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
return true;
|
|
386
|
+
}
|
|
387
|
+
function validateBucketEntry(bucket) {
|
|
388
|
+
if (!bucket || typeof bucket !== "object") {
|
|
389
|
+
return false;
|
|
390
|
+
}
|
|
391
|
+
const b = bucket;
|
|
392
|
+
if (typeof b.score !== "number" || !Number.isFinite(b.score) || b.score < 0 || b.score > 100) {
|
|
393
|
+
return false;
|
|
394
|
+
}
|
|
395
|
+
if (typeof b.rationale !== "string") {
|
|
396
|
+
return false;
|
|
397
|
+
}
|
|
398
|
+
if (!Array.isArray(b.evidence)) {
|
|
399
|
+
return false;
|
|
400
|
+
}
|
|
401
|
+
const allowEmptyArrays = JUDGE_FEATURE_FLAGS.RATIONALE_EMPTY_ARRAYS === "on";
|
|
402
|
+
if (!allowEmptyArrays && b.evidence.length === 0) {
|
|
403
|
+
return false;
|
|
404
|
+
}
|
|
405
|
+
for (const e of b.evidence) {
|
|
406
|
+
if (typeof e !== "string") {
|
|
407
|
+
return false;
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
return true;
|
|
411
|
+
}
|
|
412
|
+
function validateDimensionsEntry(dimensions) {
|
|
413
|
+
if (dimensions === void 0) {
|
|
414
|
+
return true;
|
|
415
|
+
}
|
|
416
|
+
if (!dimensions || typeof dimensions !== "object") {
|
|
417
|
+
return false;
|
|
418
|
+
}
|
|
419
|
+
const d = dimensions;
|
|
420
|
+
const allowedKeys = ["readability", "maintainability", "security", "innovation"];
|
|
421
|
+
for (const [key, value] of Object.entries(d)) {
|
|
422
|
+
if (!allowedKeys.includes(key)) {
|
|
423
|
+
return false;
|
|
424
|
+
}
|
|
425
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value < 0 || value > 100) {
|
|
426
|
+
return false;
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
return true;
|
|
430
|
+
}
|
|
431
|
+
function validateSummaryEntry(summary) {
|
|
432
|
+
if (summary === void 0) {
|
|
433
|
+
return true;
|
|
434
|
+
}
|
|
435
|
+
if (!summary || typeof summary !== "object") {
|
|
436
|
+
return false;
|
|
437
|
+
}
|
|
438
|
+
const s = summary;
|
|
439
|
+
if (s.ranking !== void 0 && !validateRankingArray(s.ranking)) {
|
|
440
|
+
return false;
|
|
441
|
+
}
|
|
442
|
+
if (s.rankingRationale !== void 0 && typeof s.rankingRationale !== "string") {
|
|
443
|
+
return false;
|
|
444
|
+
}
|
|
445
|
+
if (s.confidence !== void 0) {
|
|
446
|
+
if (typeof s.confidence !== "number" || !Number.isFinite(s.confidence) || s.confidence < 0 || s.confidence > 1) {
|
|
447
|
+
return false;
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
if (s.confidenceRationale !== void 0 && typeof s.confidenceRationale !== "string") {
|
|
451
|
+
return false;
|
|
452
|
+
}
|
|
453
|
+
if (s.top2SeparationRationale !== void 0 && typeof s.top2SeparationRationale !== "string") {
|
|
454
|
+
return false;
|
|
455
|
+
}
|
|
456
|
+
return true;
|
|
457
|
+
}
|
|
458
|
+
function validateWinnerEntry(winner) {
|
|
459
|
+
if (winner === void 0) {
|
|
460
|
+
return true;
|
|
461
|
+
}
|
|
462
|
+
if (winner === null) {
|
|
463
|
+
return true;
|
|
464
|
+
}
|
|
465
|
+
if (typeof winner !== "object") {
|
|
466
|
+
return false;
|
|
467
|
+
}
|
|
468
|
+
const w = winner;
|
|
469
|
+
const winnerId = typeof w.id === "string" && w.id.trim().length > 0 ? w.id.trim() : typeof w.variant === "string" && w.variant.trim().length > 0 ? w.variant.trim() : "";
|
|
470
|
+
if (winnerId.length === 0) {
|
|
471
|
+
return false;
|
|
472
|
+
}
|
|
473
|
+
if (w.reason !== void 0 && typeof w.reason !== "string") {
|
|
474
|
+
return false;
|
|
475
|
+
}
|
|
476
|
+
if (w.confidence !== void 0) {
|
|
477
|
+
if (typeof w.confidence !== "number" || w.confidence < 0 || w.confidence > 1) {
|
|
478
|
+
return false;
|
|
479
|
+
}
|
|
480
|
+
}
|
|
481
|
+
if (w.confidenceRationale !== void 0 && typeof w.confidenceRationale !== "string") {
|
|
482
|
+
return false;
|
|
483
|
+
}
|
|
484
|
+
if (w.top2SeparationRationale !== void 0 && typeof w.top2SeparationRationale !== "string") {
|
|
485
|
+
return false;
|
|
486
|
+
}
|
|
487
|
+
return true;
|
|
488
|
+
}
|
|
489
|
+
function validateRankingArray(value) {
|
|
490
|
+
if (!Array.isArray(value) || value.length === 0) {
|
|
491
|
+
return false;
|
|
492
|
+
}
|
|
493
|
+
const normalized = value.map((entry) => typeof entry === "string" ? entry.trim() : "");
|
|
494
|
+
if (normalized.some((entry) => entry.length === 0)) {
|
|
495
|
+
return false;
|
|
496
|
+
}
|
|
497
|
+
return new Set(normalized).size === normalized.length;
|
|
498
|
+
}
|
|
499
|
+
function validateLimitationEntry(limitation) {
|
|
500
|
+
if (limitation === void 0) {
|
|
501
|
+
return true;
|
|
502
|
+
}
|
|
503
|
+
if (!limitation || typeof limitation !== "object") {
|
|
504
|
+
return false;
|
|
505
|
+
}
|
|
506
|
+
const value = limitation;
|
|
507
|
+
if (typeof value.summary !== "string") {
|
|
508
|
+
return false;
|
|
509
|
+
}
|
|
510
|
+
if (!Array.isArray(value.evidence)) {
|
|
511
|
+
return false;
|
|
512
|
+
}
|
|
513
|
+
for (const entry of value.evidence) {
|
|
514
|
+
if (typeof entry !== "string") {
|
|
515
|
+
return false;
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
return true;
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
// src/core/judge/deliverable-gating.ts
|
|
522
|
+
function extractArtifactEvidence(variant) {
|
|
523
|
+
const filesChanged = variant.filesChanged ?? variant.gitMetrics?.filesChanged ?? variant.worktreeSummary?.filesChanged ?? variant.worktreeSummary?.statusLines?.length ?? 0;
|
|
524
|
+
const workspaceEditsApplied = variant.workspaceEditsApplied ?? 0;
|
|
525
|
+
const emptyChanges = variant.emptyChanges ?? false;
|
|
526
|
+
const hasEvidence = filesChanged > 0 && !emptyChanges || workspaceEditsApplied > 0;
|
|
527
|
+
return {
|
|
528
|
+
filesChanged,
|
|
529
|
+
workspaceEditsApplied,
|
|
530
|
+
emptyChanges,
|
|
531
|
+
hasEvidence
|
|
532
|
+
};
|
|
533
|
+
}
|
|
534
|
+
function extractDeliverableContract(enhancementSet) {
|
|
535
|
+
if (!enhancementSet) {
|
|
536
|
+
return {
|
|
537
|
+
requiresArtifact: false,
|
|
538
|
+
artifactRequiredCriteria: [],
|
|
539
|
+
hasConflict: false
|
|
540
|
+
};
|
|
541
|
+
}
|
|
542
|
+
const contractIntent = enhancementSet.intent?.category;
|
|
543
|
+
const acceptanceCriteria = enhancementSet.acceptanceCriteria ?? [];
|
|
544
|
+
const artifactRequiredCriteria = acceptanceCriteria.filter((criterion) => {
|
|
545
|
+
const verification = criterion.verification;
|
|
546
|
+
if (!verification || typeof verification !== "object") return false;
|
|
547
|
+
return verification.requiresArtifact === true;
|
|
548
|
+
}).map((criterion) => criterion.id || criterion.description);
|
|
549
|
+
const requiresArtifact = artifactRequiredCriteria.length > 0;
|
|
550
|
+
const hasConflict = requiresArtifact && contractIntent === "analyze";
|
|
551
|
+
return {
|
|
552
|
+
requiresArtifact,
|
|
553
|
+
contractIntent,
|
|
554
|
+
artifactRequiredCriteria,
|
|
555
|
+
hasConflict
|
|
556
|
+
};
|
|
557
|
+
}
|
|
558
|
+
function checkVariantCompliance(variant, contract) {
|
|
559
|
+
const evidence = extractArtifactEvidence(variant);
|
|
560
|
+
const codes = [];
|
|
561
|
+
if (!contract.requiresArtifact) {
|
|
562
|
+
return {
|
|
563
|
+
variantId: variant.variant,
|
|
564
|
+
isCompliant: true,
|
|
565
|
+
evidence,
|
|
566
|
+
codes
|
|
567
|
+
};
|
|
568
|
+
}
|
|
569
|
+
if (evidence.hasEvidence) {
|
|
570
|
+
codes.push("ARTIFACT_REQUIRED_MET");
|
|
571
|
+
return {
|
|
572
|
+
variantId: variant.variant,
|
|
573
|
+
isCompliant: true,
|
|
574
|
+
evidence,
|
|
575
|
+
codes
|
|
576
|
+
};
|
|
577
|
+
}
|
|
578
|
+
codes.push("ARTIFACT_REQUIRED_MISSING");
|
|
579
|
+
if (evidence.filesChanged > 0 && evidence.emptyChanges) {
|
|
580
|
+
codes.push("EMPTY_CHANGES_VIOLATION");
|
|
581
|
+
} else {
|
|
582
|
+
codes.push("NO_EVIDENCE_FOUND");
|
|
583
|
+
}
|
|
584
|
+
if (variant.executionIntent === "analysis" || variant.completenessScore?.category === "analysis") {
|
|
585
|
+
codes.push("VARIANT_LABEL_BYPASS");
|
|
586
|
+
}
|
|
587
|
+
if (contract.hasConflict) {
|
|
588
|
+
codes.push("ANALYSIS_INTENT_CONFLICT");
|
|
589
|
+
}
|
|
590
|
+
return {
|
|
591
|
+
variantId: variant.variant,
|
|
592
|
+
isCompliant: false,
|
|
593
|
+
evidence,
|
|
594
|
+
codes,
|
|
595
|
+
reason: `Missing required artifacts for criteria: ${contract.artifactRequiredCriteria.join(", ")}`
|
|
596
|
+
};
|
|
597
|
+
}
|
|
598
|
+
function filterByDeliverableCompliance(variants, enhancementSet, options = {}) {
|
|
599
|
+
const { skipManualVerification = true } = options;
|
|
600
|
+
void options.mode;
|
|
601
|
+
const contract = extractDeliverableContract(enhancementSet);
|
|
602
|
+
if (!contract.requiresArtifact) {
|
|
603
|
+
return {
|
|
604
|
+
filtered: variants,
|
|
605
|
+
compliance: variants.map((v) => checkVariantCompliance(v, contract)),
|
|
606
|
+
hadToRelax: false
|
|
607
|
+
};
|
|
608
|
+
}
|
|
609
|
+
if (skipManualVerification) {
|
|
610
|
+
const hasManualVerification = enhancementSet?.acceptanceCriteria?.some(
|
|
611
|
+
(criterion) => criterion.verification?.method === "manual"
|
|
612
|
+
);
|
|
613
|
+
if (hasManualVerification) {
|
|
614
|
+
return {
|
|
615
|
+
filtered: variants,
|
|
616
|
+
compliance: variants.map((v) => checkVariantCompliance(v, contract)),
|
|
617
|
+
hadToRelax: false,
|
|
618
|
+
warnings: ["Deliverable gating skipped: manual verification required"]
|
|
619
|
+
};
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
const compliance = variants.map((v) => checkVariantCompliance(v, contract));
|
|
623
|
+
const nonCompliantCount = compliance.filter((c) => !c.isCompliant).length;
|
|
624
|
+
const warnings = [];
|
|
625
|
+
if (nonCompliantCount > 0) {
|
|
626
|
+
warnings.push(
|
|
627
|
+
`${nonCompliantCount} variant(s) missing deliverable requirements: ${contract.artifactRequiredCriteria.join(", ")}`
|
|
628
|
+
);
|
|
629
|
+
}
|
|
630
|
+
if (contract.hasConflict) {
|
|
631
|
+
warnings.push('Contract conflict: intent is "analysis" but artifacts are required');
|
|
632
|
+
}
|
|
633
|
+
compliance.forEach((c) => {
|
|
634
|
+
if (!c.isCompliant) {
|
|
635
|
+
c.codes.push("SOFT_MODE_WARNING");
|
|
636
|
+
}
|
|
637
|
+
});
|
|
638
|
+
return {
|
|
639
|
+
filtered: variants,
|
|
640
|
+
compliance,
|
|
641
|
+
hadToRelax: warnings.length > 0,
|
|
642
|
+
...warnings.length > 0 ? { warnings } : {}
|
|
643
|
+
};
|
|
644
|
+
}
|
|
645
|
+
function getComplianceTelemetry(compliance) {
|
|
646
|
+
const codes = {
|
|
647
|
+
ARTIFACT_REQUIRED_MET: 0,
|
|
648
|
+
ARTIFACT_REQUIRED_MISSING: 0,
|
|
649
|
+
ANALYSIS_INTENT_CONFLICT: 0,
|
|
650
|
+
EMPTY_CHANGES_VIOLATION: 0,
|
|
651
|
+
NO_EVIDENCE_FOUND: 0,
|
|
652
|
+
VARIANT_LABEL_BYPASS: 0,
|
|
653
|
+
SOFT_MODE_WARNING: 0,
|
|
654
|
+
HARD_MODE_FAILURE: 0
|
|
655
|
+
};
|
|
656
|
+
let compliantCount = 0;
|
|
657
|
+
let nonCompliantCount = 0;
|
|
658
|
+
for (const result of compliance) {
|
|
659
|
+
if (result.isCompliant) {
|
|
660
|
+
compliantCount++;
|
|
661
|
+
} else {
|
|
662
|
+
nonCompliantCount++;
|
|
663
|
+
}
|
|
664
|
+
for (const code of result.codes) {
|
|
665
|
+
codes[code]++;
|
|
666
|
+
}
|
|
667
|
+
}
|
|
668
|
+
return {
|
|
669
|
+
totalVariants: compliance.length,
|
|
670
|
+
compliantCount,
|
|
671
|
+
nonCompliantCount,
|
|
672
|
+
codes,
|
|
673
|
+
hadLabelBypass: codes.VARIANT_LABEL_BYPASS > 0,
|
|
674
|
+
hadConflict: codes.ANALYSIS_INTENT_CONFLICT > 0
|
|
675
|
+
};
|
|
676
|
+
}
|
|
677
|
+
|
|
678
|
+
// src/core/judge/judge-llm-call.ts
|
|
679
|
+
async function executeWithSchemaRepair(args) {
|
|
680
|
+
const llmStart = Date.now();
|
|
681
|
+
let rawResponse;
|
|
682
|
+
let provider;
|
|
683
|
+
let model;
|
|
684
|
+
let thinkingBlocksObserved;
|
|
685
|
+
let tokenUsage;
|
|
686
|
+
let totalCostUsd;
|
|
687
|
+
let sandboxProvenance;
|
|
688
|
+
let sandboxEvidence;
|
|
689
|
+
try {
|
|
690
|
+
const result = await args.executor.evaluate(args.prompt, args.execOptions);
|
|
691
|
+
rawResponse = result.output;
|
|
692
|
+
provider = result.provider;
|
|
693
|
+
model = result.model;
|
|
694
|
+
thinkingBlocksObserved = result.thinkingBlocksObserved;
|
|
695
|
+
tokenUsage = result.tokenUsage;
|
|
696
|
+
totalCostUsd = result.totalCostUsd;
|
|
697
|
+
sandboxProvenance = result.sandboxProvenance;
|
|
698
|
+
sandboxEvidence = result.sandboxEvidence;
|
|
699
|
+
} catch (error) {
|
|
700
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
701
|
+
throw new Error(`${args.errorPrefix}: Provider call failed - ${message}`);
|
|
702
|
+
}
|
|
703
|
+
let parsed = args.parse(rawResponse);
|
|
704
|
+
let schemaRepairRetry;
|
|
705
|
+
if (!parsed || !args.validate(parsed)) {
|
|
706
|
+
const invalidStandaloneJson = parsed ? args.isSuspectParse?.(rawResponse) ?? false : false;
|
|
707
|
+
if (!args.strict) {
|
|
708
|
+
const firstFailReason = !parsed || invalidStandaloneJson ? args.classifyFailure(rawResponse) : "schema_validation_failed";
|
|
709
|
+
schemaRepairRetry = {
|
|
710
|
+
attempted: true,
|
|
711
|
+
succeeded: false,
|
|
712
|
+
firstFailureReason: firstFailReason
|
|
713
|
+
};
|
|
714
|
+
args.persistArtifact(rawResponse, `${firstFailReason}_before_retry`);
|
|
715
|
+
args.onRetryWarn(firstFailReason);
|
|
716
|
+
const retryPrompt = args.prompt + args.formatReminder;
|
|
717
|
+
let retryResponse;
|
|
718
|
+
try {
|
|
719
|
+
const retryResult = await args.executor.evaluate(retryPrompt, args.execOptions);
|
|
720
|
+
retryResponse = retryResult.output;
|
|
721
|
+
provider = retryResult.provider;
|
|
722
|
+
model = retryResult.model;
|
|
723
|
+
thinkingBlocksObserved = retryResult.thinkingBlocksObserved;
|
|
724
|
+
tokenUsage = mergeTokenUsage(tokenUsage, retryResult.tokenUsage);
|
|
725
|
+
totalCostUsd = (totalCostUsd ?? 0) + (typeof retryResult.totalCostUsd === "number" ? retryResult.totalCostUsd : 0);
|
|
726
|
+
sandboxProvenance = retryResult.sandboxProvenance;
|
|
727
|
+
sandboxEvidence = retryResult.sandboxEvidence;
|
|
728
|
+
} catch (retryError) {
|
|
729
|
+
const message = retryError instanceof Error ? retryError.message : String(retryError);
|
|
730
|
+
throw new Error(`${args.errorPrefix}: Provider call failed on schema retry - ${message}`);
|
|
731
|
+
}
|
|
732
|
+
parsed = args.parse(retryResponse);
|
|
733
|
+
if (!parsed) {
|
|
734
|
+
const retryReason = args.classifyFailure(retryResponse);
|
|
735
|
+
args.persistArtifact(retryResponse, `${retryReason}_retry`);
|
|
736
|
+
throw new Error(
|
|
737
|
+
`${args.errorPrefix}: Output parse failure after retry [reason=${retryReason}] - ${args.createResponseSnippet(retryResponse)}`
|
|
738
|
+
);
|
|
739
|
+
}
|
|
740
|
+
if (!args.validate(parsed)) {
|
|
741
|
+
args.persistArtifact(retryResponse, "schema_validation_failed_retry");
|
|
742
|
+
const detail = args.describeValidationFailure(parsed);
|
|
743
|
+
throw new Error(
|
|
744
|
+
`${args.errorPrefix}: Output validation failure after retry [reason=schema_validation_failed]${detail ? ` - ${detail}` : ""}`
|
|
745
|
+
);
|
|
746
|
+
}
|
|
747
|
+
rawResponse = retryResponse;
|
|
748
|
+
schemaRepairRetry.succeeded = true;
|
|
749
|
+
} else {
|
|
750
|
+
if (!parsed || invalidStandaloneJson) {
|
|
751
|
+
const reason = args.classifyFailure(rawResponse);
|
|
752
|
+
args.persistArtifact(rawResponse, reason);
|
|
753
|
+
throw new Error(
|
|
754
|
+
`${args.errorPrefix}: Output parse failure [reason=${reason}] - ${args.createResponseSnippet(rawResponse)}`
|
|
755
|
+
);
|
|
756
|
+
}
|
|
757
|
+
args.persistArtifact(rawResponse, "schema_validation_failed");
|
|
758
|
+
const detail = args.describeValidationFailure(parsed);
|
|
759
|
+
throw new Error(
|
|
760
|
+
`${args.errorPrefix}: Output validation failure [reason=schema_validation_failed]${detail ? ` - ${detail}` : ""}`
|
|
761
|
+
);
|
|
762
|
+
}
|
|
763
|
+
}
|
|
764
|
+
const llmMs = Date.now() - llmStart;
|
|
765
|
+
return {
|
|
766
|
+
parsed,
|
|
767
|
+
rawResponse,
|
|
768
|
+
provider,
|
|
769
|
+
model,
|
|
770
|
+
thinkingBlocksObserved,
|
|
771
|
+
tokenUsage,
|
|
772
|
+
totalCostUsd,
|
|
773
|
+
sandboxProvenance,
|
|
774
|
+
sandboxEvidence,
|
|
775
|
+
schemaRepairRetry,
|
|
776
|
+
llmMs
|
|
777
|
+
};
|
|
778
|
+
}
|
|
779
|
+
|
|
780
|
+
// src/core/judge/judge-output-schema.ts
|
|
781
|
+
function scoreParsedCandidate(parsed) {
|
|
782
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
|
|
783
|
+
return 0;
|
|
784
|
+
}
|
|
785
|
+
const record = parsed;
|
|
786
|
+
let score = 0;
|
|
787
|
+
if (Array.isArray(record.variants)) score += 50;
|
|
788
|
+
if (record.winner != null) score += 20;
|
|
789
|
+
if (record.summary != null) score += 10;
|
|
790
|
+
if (typeof record.taskType === "string") score += 5;
|
|
791
|
+
if (Array.isArray(record.criteria)) score += 50;
|
|
792
|
+
if (typeof record.version === "string") score += 10;
|
|
793
|
+
if (typeof record.variantId === "string") score += 10;
|
|
794
|
+
if (typeof record.finalScore === "number") score += 15;
|
|
795
|
+
if (Array.isArray(record.blockingIssues)) score += 15;
|
|
796
|
+
if (typeof record.ranking === "number") score += 10;
|
|
797
|
+
return score;
|
|
798
|
+
}
|
|
799
|
+
function parseJudgeOutput(text) {
|
|
800
|
+
let bestParsed = null;
|
|
801
|
+
let bestScore = Number.NEGATIVE_INFINITY;
|
|
802
|
+
for (const candidate of collectJsonCandidates(text)) {
|
|
803
|
+
const parsed = parseJsonCandidate(candidate);
|
|
804
|
+
if (parsed === null) {
|
|
805
|
+
continue;
|
|
806
|
+
}
|
|
807
|
+
const score = scoreParsedCandidate(parsed);
|
|
808
|
+
if (score > bestScore) {
|
|
809
|
+
bestParsed = parsed;
|
|
810
|
+
bestScore = score;
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
return bestParsed;
|
|
814
|
+
}
|
|
815
|
+
|
|
816
|
+
// src/core/judge/ai-judge.ts
|
|
817
|
+
var FILE_TARGET_COVERAGE_PROMPT_ITEM_LIMIT = 20;
|
|
818
|
+
function estimateTokensFromChars(chars) {
|
|
819
|
+
return Math.max(0, charsToTokens(chars));
|
|
820
|
+
}
|
|
821
|
+
function makeMetric(chars) {
|
|
822
|
+
const c = Math.max(0, Math.floor(chars));
|
|
823
|
+
return { chars: c, estimatedTokens: estimateTokensFromChars(c) };
|
|
824
|
+
}
|
|
825
|
+
var AIJudge = class {
|
|
826
|
+
executor;
|
|
827
|
+
customPrompt;
|
|
828
|
+
provider;
|
|
829
|
+
/**
|
|
830
|
+
* Label-only default model (Claude judge_primary role). Used for telemetry/labels.
|
|
831
|
+
* NOT forwarded as the judge model unless explicitly set (see `explicitModel`) so the
|
|
832
|
+
* shared judge resolver can pick a model coherent with the resolved provider.
|
|
833
|
+
*/
|
|
834
|
+
model;
|
|
835
|
+
/** Model explicitly provided at construction; undefined lets the resolver choose. */
|
|
836
|
+
explicitModel;
|
|
837
|
+
thinkingBudgetTokens;
|
|
838
|
+
requireThinking;
|
|
839
|
+
timeoutSeconds;
|
|
840
|
+
constructor(options = {}) {
|
|
841
|
+
this.executor = new LlmJudgeExecutor();
|
|
842
|
+
this.customPrompt = options.customPrompt;
|
|
843
|
+
this.provider = options.provider;
|
|
844
|
+
this.model = options.model ?? getDefaultJudgeModel();
|
|
845
|
+
this.explicitModel = options.model;
|
|
846
|
+
this.thinkingBudgetTokens = options.thinkingBudgetTokens;
|
|
847
|
+
this.requireThinking = options.requireThinking;
|
|
848
|
+
this.timeoutSeconds = options.timeoutSeconds ?? 300;
|
|
849
|
+
}
|
|
850
|
+
/**
|
|
851
|
+
* Evaluate variants using LLM-based judgment.
|
|
852
|
+
*
|
|
853
|
+
* @param variants - Array of variant results to evaluate
|
|
854
|
+
* @param prompt - Original task prompt
|
|
855
|
+
* @param options - Evaluation options
|
|
856
|
+
* @returns Judge synthesis result
|
|
857
|
+
*/
|
|
858
|
+
async evaluate(variants, prompt, options = {}) {
|
|
859
|
+
const startTime = Date.now();
|
|
860
|
+
if (!variants || variants.length === 0) {
|
|
861
|
+
return this.createEmptyResult("No variants provided for evaluation", startTime);
|
|
862
|
+
}
|
|
863
|
+
let judgeVisibleVariants = variants;
|
|
864
|
+
let compliance;
|
|
865
|
+
let gatingWarnings;
|
|
866
|
+
if (options.enhancementSet) {
|
|
867
|
+
const gatingResult = filterByDeliverableCompliance(
|
|
868
|
+
variants,
|
|
869
|
+
options.enhancementSet,
|
|
870
|
+
{ mode: "soft" }
|
|
871
|
+
// Use soft mode by default
|
|
872
|
+
);
|
|
873
|
+
compliance = gatingResult.compliance;
|
|
874
|
+
gatingWarnings = gatingResult.warnings;
|
|
875
|
+
judgeVisibleVariants = variants;
|
|
876
|
+
}
|
|
877
|
+
const evaluationContext = resolveJudgeEvaluationContext(prompt, {
|
|
878
|
+
evaluationContext: options.evaluationContext,
|
|
879
|
+
variants: judgeVisibleVariants
|
|
880
|
+
});
|
|
881
|
+
const promptTaskProfile = this.resolvePromptOnlyTaskProfile(prompt, evaluationContext);
|
|
882
|
+
const gatesStart = Date.now();
|
|
883
|
+
let preScoreGateContext;
|
|
884
|
+
try {
|
|
885
|
+
preScoreGateContext = await this.computePreScoreGateContext(
|
|
886
|
+
judgeVisibleVariants,
|
|
887
|
+
prompt,
|
|
888
|
+
options,
|
|
889
|
+
evaluationContext
|
|
890
|
+
);
|
|
891
|
+
} catch (error) {
|
|
892
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
893
|
+
console.warn(`[AIJudge] Pre-score gate context computation failed: ${message}`);
|
|
894
|
+
preScoreGateContext = {
|
|
895
|
+
gateResultsByVariant: /* @__PURE__ */ new Map(),
|
|
896
|
+
structuralRisksByVariant: /* @__PURE__ */ new Map(),
|
|
897
|
+
gateSourceByVariant: /* @__PURE__ */ new Map(),
|
|
898
|
+
isCodeTask: false
|
|
899
|
+
};
|
|
900
|
+
}
|
|
901
|
+
const preScoreGatesMs = Date.now() - gatesStart;
|
|
902
|
+
const codeTaskGuidance = this.buildCodeTaskEvaluationGuidance(
|
|
903
|
+
prompt,
|
|
904
|
+
preScoreGateContext,
|
|
905
|
+
promptTaskProfile.taskType === "code" && promptTaskProfile.confidence !== "low"
|
|
906
|
+
);
|
|
907
|
+
const enrichedJudgeVisibleVariants = preScoreGateContext.enrichedVariants ?? judgeVisibleVariants;
|
|
908
|
+
const anonymizeEnabled = shouldAnonymize(
|
|
909
|
+
options
|
|
910
|
+
) && enrichedJudgeVisibleVariants.length > 1;
|
|
911
|
+
let variantsForPrompt = enrichedJudgeVisibleVariants;
|
|
912
|
+
let anonymizationContext;
|
|
913
|
+
if (anonymizeEnabled) {
|
|
914
|
+
const runId = resolveAnonymizationRunId(options);
|
|
915
|
+
const taskIdSeed = typeof options.taskId === "string" && options.taskId.trim().length > 0 ? options.taskId.trim() : null;
|
|
916
|
+
const usedFallbackSeed = taskIdSeed === null;
|
|
917
|
+
const anonymizationResult = taskIdSeed ? anonymizeVariants(enrichedJudgeVisibleVariants, taskIdSeed, runId) : anonymizeVariants(
|
|
918
|
+
enrichedJudgeVisibleVariants,
|
|
919
|
+
createFallbackAnonymizationSeed(prompt, runId),
|
|
920
|
+
runId
|
|
921
|
+
);
|
|
922
|
+
variantsForPrompt = anonymizationResult.shuffledVariants;
|
|
923
|
+
anonymizationContext = {
|
|
924
|
+
result: anonymizationResult,
|
|
925
|
+
runId,
|
|
926
|
+
usedFallbackSeed
|
|
927
|
+
};
|
|
928
|
+
}
|
|
929
|
+
const resolveOriginalVariantId = (variantId) => anonymizationContext?.result.labelToVariant.get(variantId) ?? variantId;
|
|
930
|
+
const promptReviewDetection = detectExplicitReviewTask(prompt);
|
|
931
|
+
const variantHealth = observeVariantHealth({
|
|
932
|
+
variants: enrichedJudgeVisibleVariants,
|
|
933
|
+
judgeKind: "ai",
|
|
934
|
+
isExplicitReview: promptReviewDetection.isExplicitReview,
|
|
935
|
+
explicitReviewExemptsR5: true,
|
|
936
|
+
options,
|
|
937
|
+
taskId: options.taskId,
|
|
938
|
+
telemetry: options.telemetry,
|
|
939
|
+
resolveVariantId: resolveOriginalVariantId,
|
|
940
|
+
onWouldExclude: (variantId) => {
|
|
941
|
+
console.log(
|
|
942
|
+
`[AIJudge] R5 gate shadow: variant ${variantId} WOULD be skipped (no file changes, no test execution).`
|
|
943
|
+
);
|
|
944
|
+
}
|
|
945
|
+
});
|
|
946
|
+
let duplicateSignalGuidance = "";
|
|
947
|
+
if (JUDGE_FEATURE_FLAGS.DUPLICATE_CONTENT_SIGNAL !== "off") {
|
|
948
|
+
const duplicateSignals = variantsForPrompt.map((variant) => {
|
|
949
|
+
const scan = detectDuplicateContent(extractDiffText(variant) || variant.output || "");
|
|
950
|
+
return {
|
|
951
|
+
// This object also drives promptDelta telemetry, so record the identifier
|
|
952
|
+
// exactly as shown to the judge rather than leaking the original variant ID.
|
|
953
|
+
variant: anonymizationContext?.result.variantToLabel.get(variant.variant) ?? variant.variant,
|
|
954
|
+
hasDuplicates: scan.hasDuplicates,
|
|
955
|
+
duplicationRatio: scan.duplicationRatio,
|
|
956
|
+
duplicateBlockCount: scan.duplicateBlockCount
|
|
957
|
+
};
|
|
958
|
+
});
|
|
959
|
+
if (options.telemetry?.logShadowDelta && options.taskId?.trim()) {
|
|
960
|
+
options.telemetry.logShadowDelta({
|
|
961
|
+
tier: "prompt",
|
|
962
|
+
flag: "DUPLICATE_CONTENT_SIGNAL",
|
|
963
|
+
taskId: options.taskId.trim(),
|
|
964
|
+
applied: JUDGE_FEATURE_FLAGS.DUPLICATE_CONTENT_SIGNAL === "on",
|
|
965
|
+
addedInstructions: JUDGE_FEATURE_FLAGS.DUPLICATE_CONTENT_SIGNAL === "on" ? 1 : 0,
|
|
966
|
+
promptDelta: {
|
|
967
|
+
duplicateSignals
|
|
968
|
+
}
|
|
969
|
+
});
|
|
970
|
+
}
|
|
971
|
+
if (JUDGE_FEATURE_FLAGS.DUPLICATE_CONTENT_SIGNAL === "on") {
|
|
972
|
+
duplicateSignalGuidance = [
|
|
973
|
+
"Duplicate content signal (pre-LLM factual context):",
|
|
974
|
+
...duplicateSignals.map((signal) => {
|
|
975
|
+
const ratioPercent = Math.round(signal.duplicationRatio * 1e3) / 10;
|
|
976
|
+
return `- ${signal.variant}: duplicationRatio=${ratioPercent}%, hasDuplicates=${signal.hasDuplicates}, duplicateBlocks=${signal.duplicateBlockCount}`;
|
|
977
|
+
}),
|
|
978
|
+
"Penalize unnecessary duplicated blocks that do not improve correctness or delivery."
|
|
979
|
+
].join("\n");
|
|
980
|
+
}
|
|
981
|
+
}
|
|
982
|
+
const combinedGuidanceBase = codeTaskGuidance ? `${promptTaskProfile.guidance}
|
|
983
|
+
|
|
984
|
+
${codeTaskGuidance}` : promptTaskProfile.guidance;
|
|
985
|
+
const combinedGuidance = duplicateSignalGuidance ? `${combinedGuidanceBase}
|
|
986
|
+
|
|
987
|
+
${duplicateSignalGuidance}` : combinedGuidanceBase;
|
|
988
|
+
const {
|
|
989
|
+
context: promptContext,
|
|
990
|
+
evidenceMeta,
|
|
991
|
+
variantLabelMapping,
|
|
992
|
+
variantMeasurements
|
|
993
|
+
} = this.buildPromptContext(
|
|
994
|
+
variantsForPrompt,
|
|
995
|
+
prompt,
|
|
996
|
+
combinedGuidance,
|
|
997
|
+
preScoreGateContext,
|
|
998
|
+
anonymizationContext?.result.variantToLabel,
|
|
999
|
+
evaluationContext,
|
|
1000
|
+
options.variantPaths
|
|
1001
|
+
);
|
|
1002
|
+
if (options.judgeProfile) {
|
|
1003
|
+
promptContext.judgeProfile = options.judgeProfile;
|
|
1004
|
+
}
|
|
1005
|
+
const fullPrompt = buildAIJudgePrompt(promptContext, this.customPrompt);
|
|
1006
|
+
const promptTelemetry = this.computeStandalonePromptTelemetry(
|
|
1007
|
+
fullPrompt,
|
|
1008
|
+
promptContext,
|
|
1009
|
+
variantMeasurements
|
|
1010
|
+
);
|
|
1011
|
+
const callResult = await executeWithSchemaRepair({
|
|
1012
|
+
executor: this.executor,
|
|
1013
|
+
prompt: fullPrompt,
|
|
1014
|
+
execOptions: {
|
|
1015
|
+
provider: this.provider,
|
|
1016
|
+
// Explicit model only (per-call override > constructor explicit); undefined lets
|
|
1017
|
+
// the shared judge resolver pick a model coherent with the resolved provider for
|
|
1018
|
+
// the 'ai' tier. Label/telemetry fall back to this.model elsewhere.
|
|
1019
|
+
model: options.model ?? this.explicitModel,
|
|
1020
|
+
judgeType: "ai",
|
|
1021
|
+
timeoutSeconds: options.timeoutSeconds ?? this.timeoutSeconds,
|
|
1022
|
+
maxAttempts: options.maxAttempts,
|
|
1023
|
+
thinkingBudgetTokens: this.thinkingBudgetTokens,
|
|
1024
|
+
requireThinking: this.requireThinking,
|
|
1025
|
+
variantCount: variantsForPrompt.length,
|
|
1026
|
+
env: options.env,
|
|
1027
|
+
deadlineMs: options.deadlineMs,
|
|
1028
|
+
abortSignal: options.abortSignal,
|
|
1029
|
+
progressWriter: options.progressWriter,
|
|
1030
|
+
progressLabel: options.progressLabel
|
|
1031
|
+
},
|
|
1032
|
+
parse: (raw) => this.repairMissingOverallScores(parseJudgeOutput(raw)),
|
|
1033
|
+
validate: (value) => validateAIJudgeOutput(value),
|
|
1034
|
+
formatReminder: "\n\nYour previous response was not valid JSON matching the required schema. Please respond with ONLY a valid JSON object matching the specified format.",
|
|
1035
|
+
classifyFailure: (raw) => this.classifyParseFailureReason(raw),
|
|
1036
|
+
isSuspectParse: (raw) => isStandaloneMalformedJsonCandidate(raw),
|
|
1037
|
+
strict: options.maxAttempts === 1,
|
|
1038
|
+
errorPrefix: "AIJudge",
|
|
1039
|
+
persistArtifact: (response, label) => this.persistFailedResponse(response, options.taskId, label),
|
|
1040
|
+
onRetryWarn: (reason) => {
|
|
1041
|
+
console.warn(
|
|
1042
|
+
`[AIJudge] Output parse/schema failure (reason=${reason}) \u2014 retrying once with format reminder`
|
|
1043
|
+
);
|
|
1044
|
+
},
|
|
1045
|
+
createResponseSnippet: (response) => this.createResponseSnippet(response),
|
|
1046
|
+
describeValidationFailure: (value) => this.describeValidationFailure(value)
|
|
1047
|
+
});
|
|
1048
|
+
const llmProvider = callResult.provider;
|
|
1049
|
+
const llmModel = callResult.model;
|
|
1050
|
+
const thinkingBlocksObserved = callResult.thinkingBlocksObserved;
|
|
1051
|
+
const llmTokenUsage = callResult.tokenUsage;
|
|
1052
|
+
const llmTotalCostUsd = callResult.totalCostUsd;
|
|
1053
|
+
const llmSandboxProvenance = callResult.sandboxProvenance;
|
|
1054
|
+
const llmSandboxEvidence = callResult.sandboxEvidence;
|
|
1055
|
+
const schemaRepairRetry = callResult.schemaRepairRetry;
|
|
1056
|
+
const llmMs = callResult.llmMs;
|
|
1057
|
+
const parsedOutput = callResult.parsed;
|
|
1058
|
+
const normalizedOutput = this.normalizeAnonymizedOutput(
|
|
1059
|
+
parsedOutput,
|
|
1060
|
+
anonymizationContext?.result.labelToVariant
|
|
1061
|
+
);
|
|
1062
|
+
const evidenceUsage = this.computeEvidenceUsageMetrics(
|
|
1063
|
+
parsedOutput,
|
|
1064
|
+
variantsForPrompt,
|
|
1065
|
+
evidenceMeta
|
|
1066
|
+
);
|
|
1067
|
+
const agenticOutput = parsedOutput.agentic;
|
|
1068
|
+
const synthesis = this.mapToSynthesisResult(
|
|
1069
|
+
normalizedOutput,
|
|
1070
|
+
variantsForPrompt,
|
|
1071
|
+
startTime,
|
|
1072
|
+
preScoreGateContext
|
|
1073
|
+
);
|
|
1074
|
+
if (synthesis.reasoning) {
|
|
1075
|
+
const labelMapping = anonymizationContext?.result.labelToVariant ? Object.fromEntries(anonymizationContext.result.labelToVariant) : void 0;
|
|
1076
|
+
synthesis.reasoning = deAnonymizeRationale(
|
|
1077
|
+
synthesis.reasoning,
|
|
1078
|
+
variantLabelMapping,
|
|
1079
|
+
labelMapping
|
|
1080
|
+
);
|
|
1081
|
+
}
|
|
1082
|
+
synthesis.variantHealth = variantHealth;
|
|
1083
|
+
const audit = synthesis.metadata.audit ?? {};
|
|
1084
|
+
const auditPayload = {
|
|
1085
|
+
...audit,
|
|
1086
|
+
evidenceUsage,
|
|
1087
|
+
promptTelemetry,
|
|
1088
|
+
llmProvider: llmProvider ?? null,
|
|
1089
|
+
thinkingBlocksObserved: thinkingBlocksObserved ?? null,
|
|
1090
|
+
tokenUsage: llmTokenUsage ?? null,
|
|
1091
|
+
aiTokensUsed: sumTokenUsage(llmTokenUsage) ?? null,
|
|
1092
|
+
aiCostUsd: typeof llmTotalCostUsd === "number" ? llmTotalCostUsd : null,
|
|
1093
|
+
sandboxProvenance: llmSandboxProvenance ?? null,
|
|
1094
|
+
sandboxEvidence: llmSandboxEvidence ?? null,
|
|
1095
|
+
routing: normalizedOutput.routing ?? null,
|
|
1096
|
+
rankingConfidenceRationale: normalizedOutput.summary?.confidenceRationale ?? normalizedOutput.winner?.confidenceRationale ?? null,
|
|
1097
|
+
top2SeparationRationale: normalizedOutput.summary?.top2SeparationRationale ?? normalizedOutput.winner?.top2SeparationRationale ?? null,
|
|
1098
|
+
confidenceGrounding: {
|
|
1099
|
+
confidenceRationalePresent: Boolean(
|
|
1100
|
+
normalizedOutput.summary?.confidenceRationale?.trim() ?? normalizedOutput.winner?.confidenceRationale?.trim()
|
|
1101
|
+
),
|
|
1102
|
+
top2SeparationPresent: Boolean(
|
|
1103
|
+
normalizedOutput.summary?.top2SeparationRationale?.trim() ?? normalizedOutput.winner?.top2SeparationRationale?.trim()
|
|
1104
|
+
)
|
|
1105
|
+
},
|
|
1106
|
+
variantLabelMapping,
|
|
1107
|
+
taskProfile: {
|
|
1108
|
+
taskType: promptTaskProfile.taskType,
|
|
1109
|
+
confidence: promptTaskProfile.confidence,
|
|
1110
|
+
source: promptTaskProfile.source,
|
|
1111
|
+
limitedEmpiricalSignals: promptTaskProfile.limitedEmpiricalSignals,
|
|
1112
|
+
reasoning: promptTaskProfile.reasoning ?? null
|
|
1113
|
+
},
|
|
1114
|
+
timingMs: {
|
|
1115
|
+
preScoreGatesMs,
|
|
1116
|
+
llmMs
|
|
1117
|
+
},
|
|
1118
|
+
schemaRepairRetry: schemaRepairRetry ?? {
|
|
1119
|
+
attempted: false,
|
|
1120
|
+
succeeded: false
|
|
1121
|
+
}
|
|
1122
|
+
};
|
|
1123
|
+
if (anonymizationContext) {
|
|
1124
|
+
auditPayload.anonymization = {
|
|
1125
|
+
enabled: true,
|
|
1126
|
+
judgeType: "ai",
|
|
1127
|
+
seed: anonymizationContext.result.seed,
|
|
1128
|
+
runId: anonymizationContext.runId ?? null,
|
|
1129
|
+
usedFallbackSeed: anonymizationContext.usedFallbackSeed,
|
|
1130
|
+
originalOrder: judgeVisibleVariants.map((variant) => variant.variant),
|
|
1131
|
+
shuffledOrder: anonymizationContext.result.shuffledVariants.map(
|
|
1132
|
+
(variant) => variant.variant
|
|
1133
|
+
),
|
|
1134
|
+
labelsInOrder: variantsForPrompt.map(
|
|
1135
|
+
(variant) => anonymizationContext.result.variantToLabel.get(variant.variant) ?? variant.variant
|
|
1136
|
+
),
|
|
1137
|
+
variantIdMapping: variantsForPrompt.map((variant, index) => ({
|
|
1138
|
+
id: `v${index + 1}`,
|
|
1139
|
+
variant: variant.variant,
|
|
1140
|
+
label: anonymizationContext.result.variantToLabel.get(variant.variant) ?? null
|
|
1141
|
+
})),
|
|
1142
|
+
labelToVariant: Object.fromEntries(anonymizationContext.result.labelToVariant)
|
|
1143
|
+
};
|
|
1144
|
+
}
|
|
1145
|
+
if (agenticOutput !== void 0) {
|
|
1146
|
+
auditPayload.agenticOutput = agenticOutput;
|
|
1147
|
+
}
|
|
1148
|
+
synthesis.metadata.audit = auditPayload;
|
|
1149
|
+
if (llmModel) {
|
|
1150
|
+
synthesis.metadata.model = llmModel;
|
|
1151
|
+
}
|
|
1152
|
+
if (compliance) {
|
|
1153
|
+
const telemetry = getComplianceTelemetry(compliance);
|
|
1154
|
+
const existingAudit = synthesis.metadata.audit ?? {};
|
|
1155
|
+
synthesis.metadata.audit = {
|
|
1156
|
+
...existingAudit,
|
|
1157
|
+
deliverableCompliance: telemetry,
|
|
1158
|
+
gatingWarnings
|
|
1159
|
+
};
|
|
1160
|
+
}
|
|
1161
|
+
this.validateFactualConsistency(
|
|
1162
|
+
normalizedOutput,
|
|
1163
|
+
variantsForPrompt,
|
|
1164
|
+
preScoreGateContext,
|
|
1165
|
+
synthesis
|
|
1166
|
+
);
|
|
1167
|
+
return synthesis;
|
|
1168
|
+
}
|
|
1169
|
+
async computePreScoreGateContext(variants, taskPrompt, options, evaluationContext) {
|
|
1170
|
+
let resolution = evaluationContext?.taskTypeResolution;
|
|
1171
|
+
if (!resolution) {
|
|
1172
|
+
try {
|
|
1173
|
+
const spec = parsePromptSpec(taskPrompt);
|
|
1174
|
+
const changedFiles = Array.from(
|
|
1175
|
+
new Set(
|
|
1176
|
+
variants.flatMap((variant) => {
|
|
1177
|
+
try {
|
|
1178
|
+
return extractChangedFilesMeta(extractDiffText(variant)).map((meta) => meta.path);
|
|
1179
|
+
} catch {
|
|
1180
|
+
return [];
|
|
1181
|
+
}
|
|
1182
|
+
})
|
|
1183
|
+
)
|
|
1184
|
+
);
|
|
1185
|
+
resolution = resolveTaskType(taskPrompt, {
|
|
1186
|
+
fileTargets: spec.fileConstraint.targets,
|
|
1187
|
+
changedFiles
|
|
1188
|
+
});
|
|
1189
|
+
} catch {
|
|
1190
|
+
resolution = void 0;
|
|
1191
|
+
}
|
|
1192
|
+
}
|
|
1193
|
+
return computeDeterministicPreScoreGateContext(
|
|
1194
|
+
variants,
|
|
1195
|
+
taskPrompt,
|
|
1196
|
+
options,
|
|
1197
|
+
"AIJudge",
|
|
1198
|
+
resolution
|
|
1199
|
+
);
|
|
1200
|
+
}
|
|
1201
|
+
buildArtifactScopeEvidenceSection(variant, fileConstraint) {
|
|
1202
|
+
const inventory = buildVariantFileInventory(variant, { fileConstraint });
|
|
1203
|
+
if (inventory.totalArtifactsChanged === 0) return void 0;
|
|
1204
|
+
const lines = [
|
|
1205
|
+
`Artifacts changed: ${inventory.totalArtifactsChanged}`,
|
|
1206
|
+
`Scope hygiene risk: ${inventory.scopeHygieneRisk}`,
|
|
1207
|
+
"Artifact list:",
|
|
1208
|
+
...inventory.artifactList.map((artifact) => `- ${artifact}`)
|
|
1209
|
+
];
|
|
1210
|
+
const coverage = inventory.fileTargetCoverage;
|
|
1211
|
+
if (coverage) {
|
|
1212
|
+
lines.push(`FILE target coverage mode: ${coverage.mode}`);
|
|
1213
|
+
lines.push(`FILE targets touched: ${coverage.touchedTargets.length}/${coverage.targetCount}`);
|
|
1214
|
+
if (coverage.touchedTargets.length > 0) {
|
|
1215
|
+
lines.push("Touched FILE targets:");
|
|
1216
|
+
lines.push(...this.formatBoundedCoverageItems(coverage.touchedTargets));
|
|
1217
|
+
}
|
|
1218
|
+
if (coverage.missingTargets.length > 0) {
|
|
1219
|
+
lines.push("Missing FILE targets:");
|
|
1220
|
+
lines.push(...this.formatBoundedCoverageItems(coverage.missingTargets));
|
|
1221
|
+
}
|
|
1222
|
+
if (coverage.offTargetArtifacts.length > 0) {
|
|
1223
|
+
lines.push("Off-target artifacts for hard FILE scope:");
|
|
1224
|
+
lines.push(...this.formatBoundedCoverageItems(coverage.offTargetArtifacts));
|
|
1225
|
+
}
|
|
1226
|
+
}
|
|
1227
|
+
if (inventory.suspiciousArtifacts.length > 0) {
|
|
1228
|
+
lines.push("Suspicious/unmanaged/generated/state artifacts:");
|
|
1229
|
+
lines.push(...inventory.suspiciousArtifacts.map((artifact) => `- ${artifact}`));
|
|
1230
|
+
}
|
|
1231
|
+
return lines.join("\n");
|
|
1232
|
+
}
|
|
1233
|
+
formatBoundedCoverageItems(items) {
|
|
1234
|
+
const shown = items.slice(0, FILE_TARGET_COVERAGE_PROMPT_ITEM_LIMIT).map((item) => `- ${item}`);
|
|
1235
|
+
const omitted = items.length - shown.length;
|
|
1236
|
+
if (omitted > 0) {
|
|
1237
|
+
shown.push(`- ... ${omitted} more`);
|
|
1238
|
+
}
|
|
1239
|
+
return shown;
|
|
1240
|
+
}
|
|
1241
|
+
buildDeterministicPreScoreGateSection(gateResults, structuralRisks, options) {
|
|
1242
|
+
return buildDeterministicPreScoreGateSection(gateResults, structuralRisks, options);
|
|
1243
|
+
}
|
|
1244
|
+
normalizeAnonymizedOutput(output, labelToVariant) {
|
|
1245
|
+
if (!labelToVariant || labelToVariant.size === 0) return output;
|
|
1246
|
+
const normalizeVariantToken = (value) => {
|
|
1247
|
+
if (typeof value !== "string") return value;
|
|
1248
|
+
const trimmed = value.trim();
|
|
1249
|
+
if (trimmed.length === 0) return value;
|
|
1250
|
+
return labelToVariant.get(trimmed) ?? value;
|
|
1251
|
+
};
|
|
1252
|
+
return {
|
|
1253
|
+
...output,
|
|
1254
|
+
variants: output.variants.map((variant) => ({
|
|
1255
|
+
...variant,
|
|
1256
|
+
id: normalizeVariantToken(variant.id) ?? variant.id
|
|
1257
|
+
})),
|
|
1258
|
+
ranking: Array.isArray(output.ranking) ? output.ranking.map((entry) => normalizeVariantToken(entry) ?? entry) : output.ranking,
|
|
1259
|
+
summary: output.summary ? {
|
|
1260
|
+
...output.summary,
|
|
1261
|
+
ranking: Array.isArray(output.summary.ranking) ? output.summary.ranking.map((entry) => normalizeVariantToken(entry) ?? entry) : output.summary.ranking
|
|
1262
|
+
} : output.summary,
|
|
1263
|
+
winner: output.winner ? {
|
|
1264
|
+
...output.winner,
|
|
1265
|
+
variant: typeof output.winner.variant === "string" ? normalizeVariantToken(output.winner.variant) ?? output.winner.variant : output.winner.variant,
|
|
1266
|
+
id: typeof output.winner.id === "string" ? normalizeVariantToken(output.winner.id) ?? output.winner.id : output.winner.id
|
|
1267
|
+
} : output.winner
|
|
1268
|
+
};
|
|
1269
|
+
}
|
|
1270
|
+
/**
|
|
1271
|
+
* Build the prompt context for AI evaluation.
|
|
1272
|
+
*/
|
|
1273
|
+
buildPromptContext(variants, taskPrompt, taskProfileGuidance, preScoreGateContext, variantToLabel, evaluationContext, worktreePaths) {
|
|
1274
|
+
const evidenceMeta = [];
|
|
1275
|
+
const variantLabelMapping = {};
|
|
1276
|
+
const variantMeasurements = {};
|
|
1277
|
+
const promptSpec = parsePromptSpec(taskPrompt);
|
|
1278
|
+
const sharedTaskType = evaluationContext?.taskTypeResolution.taskType;
|
|
1279
|
+
const variantSections = variants.map((variant, index) => {
|
|
1280
|
+
const promptVariantId = `v${index + 1}`;
|
|
1281
|
+
variantLabelMapping[promptVariantId] = variant.variant;
|
|
1282
|
+
const deliverableExcerpt = buildVariantDeliverableExcerptForJudge(variant).text;
|
|
1283
|
+
const variantDisplayName = variantToLabel?.get(variant.variant) ?? variant.variant;
|
|
1284
|
+
const promptOrderMode = JUDGE_FEATURE_FLAGS.JUDGE_PROFILE;
|
|
1285
|
+
const evidenceList = buildVariantEvidenceCandidatesForJudge(variant, {
|
|
1286
|
+
maxEvidenceCandidates: computePromptOrderEvidenceCandidateLimit(20, promptOrderMode)
|
|
1287
|
+
});
|
|
1288
|
+
evidenceMeta.push({
|
|
1289
|
+
variantId: promptVariantId,
|
|
1290
|
+
variantName: variantDisplayName,
|
|
1291
|
+
evidenceCandidateCount: evidenceList.length
|
|
1292
|
+
});
|
|
1293
|
+
let section = formatVariantSection(
|
|
1294
|
+
promptVariantId,
|
|
1295
|
+
variantDisplayName,
|
|
1296
|
+
evidenceList,
|
|
1297
|
+
deliverableExcerpt
|
|
1298
|
+
);
|
|
1299
|
+
const verificationProvenance = formatVerificationProvenanceForPrompt(variant, {
|
|
1300
|
+
worktreePath: worktreePaths?.[variant.variant]
|
|
1301
|
+
});
|
|
1302
|
+
if (verificationProvenance) {
|
|
1303
|
+
section += `
|
|
1304
|
+
|
|
1305
|
+
${verificationProvenance}`;
|
|
1306
|
+
}
|
|
1307
|
+
const artifactScopeEvidence = this.buildArtifactScopeEvidenceSection(
|
|
1308
|
+
variant,
|
|
1309
|
+
promptSpec.fileConstraint
|
|
1310
|
+
);
|
|
1311
|
+
if (artifactScopeEvidence) {
|
|
1312
|
+
section += `
|
|
1313
|
+
|
|
1314
|
+
### Artifact Scope Evidence
|
|
1315
|
+
${artifactScopeEvidence}`;
|
|
1316
|
+
}
|
|
1317
|
+
if (preScoreGateContext?.isCodeTask) {
|
|
1318
|
+
const gateSource = preScoreGateContext.gateSourceByVariant?.get(variant.variant);
|
|
1319
|
+
const gateSection = this.buildDeterministicPreScoreGateSection(
|
|
1320
|
+
preScoreGateContext.gateResultsByVariant.get(variant.variant),
|
|
1321
|
+
preScoreGateContext.structuralRisksByVariant.get(variant.variant),
|
|
1322
|
+
{ isCodeTask: true, gateSource }
|
|
1323
|
+
);
|
|
1324
|
+
if (gateSection) {
|
|
1325
|
+
section += `
|
|
1326
|
+
|
|
1327
|
+
### Pre-Score Gate Data
|
|
1328
|
+
${gateSection}`;
|
|
1329
|
+
}
|
|
1330
|
+
}
|
|
1331
|
+
const evidenceSectionText = evidenceList.length > 0 ? [
|
|
1332
|
+
"**Evidence Candidates** (cite as E1, E2, etc.):",
|
|
1333
|
+
...evidenceList.map((e, i) => `E${i + 1}: ${e}`),
|
|
1334
|
+
""
|
|
1335
|
+
].join("\n") : "";
|
|
1336
|
+
variantMeasurements[variant.variant] = {
|
|
1337
|
+
deliverableExcerptChars: deliverableExcerpt.length,
|
|
1338
|
+
evidenceCandidatesChars: evidenceSectionText.length,
|
|
1339
|
+
variantBlockChars: section.length
|
|
1340
|
+
};
|
|
1341
|
+
return section;
|
|
1342
|
+
}).join("\n");
|
|
1343
|
+
const taskTypeLine = sharedTaskType ? `**TASK TYPE**: ${sharedTaskType}` : "";
|
|
1344
|
+
const guidanceWithTaskType = taskTypeLine ? `${taskTypeLine}
|
|
1345
|
+
|
|
1346
|
+
${taskProfileGuidance}` : taskProfileGuidance;
|
|
1347
|
+
return {
|
|
1348
|
+
context: {
|
|
1349
|
+
taskPrompt,
|
|
1350
|
+
variantSections,
|
|
1351
|
+
taskProfileGuidance: guidanceWithTaskType
|
|
1352
|
+
},
|
|
1353
|
+
evidenceMeta,
|
|
1354
|
+
variantLabelMapping,
|
|
1355
|
+
variantMeasurements
|
|
1356
|
+
};
|
|
1357
|
+
}
|
|
1358
|
+
/**
|
|
1359
|
+
* Compute prompt telemetry from the standalone AI judge prompt path.
|
|
1360
|
+
*
|
|
1361
|
+
* Measures section sizes in the already-built prompt without mutating it.
|
|
1362
|
+
* Output shape is compatible with extractPromptTelemetrySummaryFromAudit().
|
|
1363
|
+
*/
|
|
1364
|
+
computeStandalonePromptTelemetry(fullPrompt, context, variantMeasurements) {
|
|
1365
|
+
const totalChars = fullPrompt.length;
|
|
1366
|
+
const variantBlocksChars = context.variantSections.length;
|
|
1367
|
+
const taskAndInstructionsChars = totalChars - variantBlocksChars;
|
|
1368
|
+
let totalDiffChars = 0;
|
|
1369
|
+
let totalEvidenceChars = 0;
|
|
1370
|
+
const variants = {};
|
|
1371
|
+
for (const [name, m] of Object.entries(variantMeasurements)) {
|
|
1372
|
+
totalDiffChars += m.deliverableExcerptChars;
|
|
1373
|
+
totalEvidenceChars += m.evidenceCandidatesChars;
|
|
1374
|
+
variants[name] = {
|
|
1375
|
+
deliverableExcerpt: makeMetric(m.deliverableExcerptChars),
|
|
1376
|
+
evidenceCandidatesSection: makeMetric(m.evidenceCandidatesChars),
|
|
1377
|
+
variantBlock: makeMetric(m.variantBlockChars)
|
|
1378
|
+
};
|
|
1379
|
+
}
|
|
1380
|
+
const scaffoldingChars = totalChars - totalDiffChars - totalEvidenceChars;
|
|
1381
|
+
return {
|
|
1382
|
+
estimator: "chars_div_4",
|
|
1383
|
+
path: "ai-judge-standalone",
|
|
1384
|
+
variants,
|
|
1385
|
+
sections: {
|
|
1386
|
+
taskAndInstructions: makeMetric(taskAndInstructionsChars),
|
|
1387
|
+
variantBlocks: makeMetric(variantBlocksChars),
|
|
1388
|
+
diffContent: makeMetric(totalDiffChars),
|
|
1389
|
+
evidenceCandidates: makeMetric(totalEvidenceChars),
|
|
1390
|
+
scaffoldingTotal: makeMetric(scaffoldingChars),
|
|
1391
|
+
totalPrompt: makeMetric(totalChars)
|
|
1392
|
+
}
|
|
1393
|
+
};
|
|
1394
|
+
}
|
|
1395
|
+
buildCodeTaskEvaluationGuidance(taskPrompt, preScoreGateContext, promptProfileAllowsCodeSignals) {
|
|
1396
|
+
const requiresCanonicalAuthority = /\bcanonical\s+(?:behaviou?r|rules?|contract|semantics?|source|implementation)\b/i.test(
|
|
1397
|
+
taskPrompt
|
|
1398
|
+
);
|
|
1399
|
+
const isCodeTask = preScoreGateContext?.isCodeTask === true || promptProfileAllowsCodeSignals;
|
|
1400
|
+
if (!isCodeTask) return void 0;
|
|
1401
|
+
const hasGateData = (preScoreGateContext?.gateResultsByVariant.size ?? 0) > 0;
|
|
1402
|
+
const hasStructuralData = Array.from(
|
|
1403
|
+
preScoreGateContext?.structuralRisksByVariant.values() ?? []
|
|
1404
|
+
).some((scan) => scan.findings.length > 0);
|
|
1405
|
+
if (!hasGateData && !hasStructuralData && !requiresCanonicalAuthority) return void 0;
|
|
1406
|
+
const lines = [];
|
|
1407
|
+
lines.push("## Code-Task Correctness Checklist");
|
|
1408
|
+
lines.push(
|
|
1409
|
+
"- Check for self-referential function calls without a termination condition (infinite recursion risk)."
|
|
1410
|
+
);
|
|
1411
|
+
lines.push("- Check for referenced symbols that appear undefined or missing required imports.");
|
|
1412
|
+
lines.push("- Check for return paths that conflict with stated/declared return behavior.");
|
|
1413
|
+
lines.push("- Check for error handling paths that silently swallow exceptions.");
|
|
1414
|
+
lines.push("- Check for obvious control-flow hazards (unreachable code, switch fallthrough).");
|
|
1415
|
+
lines.push("");
|
|
1416
|
+
lines.push("## Spec-Invariant & Public-Contract Check");
|
|
1417
|
+
lines.push(
|
|
1418
|
+
'Extract the explicit must/should requirements and public-contract guarantees the task states (e.g. "export the options type", "readonly results view", "validate before running", "stable ordering", "throw on invalid input"). For each variant that could affect the ranking, assess each requirement as implemented / partial / missing / unverified, citing evidence.'
|
|
1419
|
+
);
|
|
1420
|
+
lines.push(
|
|
1421
|
+
`A passing test suite is not proof a stated invariant holds: a variant's own tests may never exercise the contract (e.g. never attempt to mutate a "readonly" view, never import the type that was supposed to be exported). Judge the contract from the implementation, not from test count.`
|
|
1422
|
+
);
|
|
1423
|
+
lines.push(
|
|
1424
|
+
"Test count cuts both ways: a large test suite is not a strength when the variant's own gates are failing. When the per-variant gate data shows `buildGate.signal = fail`, `testGate.testStatus = failed` with executed tests, or new type errors, and `verificationStatus (adjusted, baseline-aware)` is neither `passed` nor `baseline_clean`, that is strong negative correctness evidence that a higher test quantity does not offset. (If adjusted verification is `baseline_clean`, treat the raw failure as pre-existing unless the diff proves otherwise.) On the strength of more tests alone, do not rank a variant whose gates fail this way above one whose gates pass; if you do, name the specific outcome evidence that overrides the gate failure."
|
|
1425
|
+
);
|
|
1426
|
+
lines.push(
|
|
1427
|
+
"Treat omission of a spec-stated public contract (exported types, readonly/immutability boundaries, validation timing, ordering guarantees) as material evidence in its own right, and do not dismiss a required contract mechanism as unnecessary complexity solely because a simpler variant omits it. If you still rank a variant that omits the contract above one that implements it, explain why the contract is not actually required here or why other evidence outweighs it."
|
|
1428
|
+
);
|
|
1429
|
+
if (requiresCanonicalAuthority) {
|
|
1430
|
+
lines.push("");
|
|
1431
|
+
lines.push("## Canonical Authority Check");
|
|
1432
|
+
lines.push("canonicalAuthority: required");
|
|
1433
|
+
lines.push(
|
|
1434
|
+
"Candidate-authored comments and tests are corroboration, not authority for what the canonical behavior requires. Ground the requirement in task-supplied or host-provided non-candidate evidence; if that authority is unavailable, mark the criterion unverified rather than accepting a candidate claim as its own proof."
|
|
1435
|
+
);
|
|
1436
|
+
}
|
|
1437
|
+
return lines.join("\n");
|
|
1438
|
+
}
|
|
1439
|
+
computeEvidenceUsageMetrics(output, variants, evidenceMeta) {
|
|
1440
|
+
const resolveEvidenceCandidateCount = (variantId) => {
|
|
1441
|
+
const exact = evidenceMeta.find(
|
|
1442
|
+
(m) => m.variantId === variantId || m.variantName === variantId
|
|
1443
|
+
);
|
|
1444
|
+
if (exact) return exact.evidenceCandidateCount;
|
|
1445
|
+
const match = variantId.match(/^v(\d+)$/i);
|
|
1446
|
+
if (match) {
|
|
1447
|
+
const index = Number(match[1]);
|
|
1448
|
+
if (Number.isFinite(index) && index >= 1 && index <= variants.length) {
|
|
1449
|
+
const byIndex = evidenceMeta.find((m) => m.variantId === `v${index}`);
|
|
1450
|
+
if (byIndex) return byIndex.evidenceCandidateCount;
|
|
1451
|
+
}
|
|
1452
|
+
}
|
|
1453
|
+
return 0;
|
|
1454
|
+
};
|
|
1455
|
+
const candidates = evidenceMeta.map((m) => m.evidenceCandidateCount);
|
|
1456
|
+
const evidenceCandidatesTotal = candidates.reduce((sum, n) => sum + n, 0);
|
|
1457
|
+
const evidenceCandidatesMin = candidates.length > 0 ? Math.min(...candidates) : 0;
|
|
1458
|
+
const evidenceCandidatesMax = candidates.length > 0 ? Math.max(...candidates) : 0;
|
|
1459
|
+
const evidenceCandidatesAvg = candidates.length > 0 ? evidenceCandidatesTotal / candidates.length : 0;
|
|
1460
|
+
const invalidSamples = [];
|
|
1461
|
+
let bucketCount = 0;
|
|
1462
|
+
let evidenceBucketsEmpty = 0;
|
|
1463
|
+
let evidenceEntriesEid = 0;
|
|
1464
|
+
let evidenceEntriesNonEid = 0;
|
|
1465
|
+
let evidenceEidInvalid = 0;
|
|
1466
|
+
for (const variant of output.variants) {
|
|
1467
|
+
const evidenceCandidateCount = resolveEvidenceCandidateCount(variant.id);
|
|
1468
|
+
const variantName = variants.find((v) => v.variant === variant.id)?.variant;
|
|
1469
|
+
const buckets = [
|
|
1470
|
+
{ bucket: "delivery", evidence: variant.delivery.evidence },
|
|
1471
|
+
{ bucket: "correctness", evidence: variant.correctness.evidence },
|
|
1472
|
+
{ bucket: "quality", evidence: variant.quality.evidence }
|
|
1473
|
+
];
|
|
1474
|
+
for (const { bucket, evidence } of buckets) {
|
|
1475
|
+
bucketCount += 1;
|
|
1476
|
+
if (evidence.length === 0) {
|
|
1477
|
+
evidenceBucketsEmpty += 1;
|
|
1478
|
+
continue;
|
|
1479
|
+
}
|
|
1480
|
+
for (const entry of evidence) {
|
|
1481
|
+
const trimmed = entry.trim();
|
|
1482
|
+
const eidMatch = trimmed.match(/^E(\d+)$/i);
|
|
1483
|
+
if (!eidMatch) {
|
|
1484
|
+
evidenceEntriesNonEid += 1;
|
|
1485
|
+
continue;
|
|
1486
|
+
}
|
|
1487
|
+
evidenceEntriesEid += 1;
|
|
1488
|
+
const eid = Number(eidMatch[1]);
|
|
1489
|
+
const invalid = !Number.isFinite(eid) || eid < 1 || eid > evidenceCandidateCount;
|
|
1490
|
+
if (invalid) {
|
|
1491
|
+
evidenceEidInvalid += 1;
|
|
1492
|
+
if (invalidSamples.length < 5) {
|
|
1493
|
+
invalidSamples.push({
|
|
1494
|
+
variantId: variant.id,
|
|
1495
|
+
variantName,
|
|
1496
|
+
bucket,
|
|
1497
|
+
evidence: trimmed,
|
|
1498
|
+
evidenceCandidateCount
|
|
1499
|
+
});
|
|
1500
|
+
}
|
|
1501
|
+
}
|
|
1502
|
+
}
|
|
1503
|
+
}
|
|
1504
|
+
}
|
|
1505
|
+
const evidenceEntriesTotal = evidenceEntriesEid + evidenceEntriesNonEid;
|
|
1506
|
+
const evidenceBucketsEmptyPct = bucketCount > 0 ? evidenceBucketsEmpty / bucketCount : 0;
|
|
1507
|
+
const evidenceEidInvalidPct = evidenceEntriesEid > 0 ? evidenceEidInvalid / evidenceEntriesEid : 0;
|
|
1508
|
+
return {
|
|
1509
|
+
policy: "prompt-only",
|
|
1510
|
+
variantsInPrompt: variants.length,
|
|
1511
|
+
variantsInOutput: output.variants.length,
|
|
1512
|
+
bucketCount,
|
|
1513
|
+
evidenceBucketsEmpty,
|
|
1514
|
+
evidenceBucketsEmptyPct,
|
|
1515
|
+
evidenceEntriesTotal,
|
|
1516
|
+
evidenceEntriesEid,
|
|
1517
|
+
evidenceEntriesNonEid,
|
|
1518
|
+
evidenceEidInvalid,
|
|
1519
|
+
evidenceEidInvalidPct,
|
|
1520
|
+
evidenceCandidatesTotal,
|
|
1521
|
+
evidenceCandidatesMin,
|
|
1522
|
+
evidenceCandidatesMax,
|
|
1523
|
+
evidenceCandidatesAvg,
|
|
1524
|
+
invalidSamples
|
|
1525
|
+
};
|
|
1526
|
+
}
|
|
1527
|
+
/**
|
|
1528
|
+
* PROMPT-ONLY task profile resolution.
|
|
1529
|
+
*
|
|
1530
|
+
* Firewall rule: this method is used only before the LLM call to shape prompt
|
|
1531
|
+
* language. Its output must never be consumed in post-parse score/winner logic.
|
|
1532
|
+
*/
|
|
1533
|
+
resolvePromptOnlyTaskProfile(taskPrompt, evaluationContext) {
|
|
1534
|
+
try {
|
|
1535
|
+
const resolution = evaluationContext?.taskTypeResolution ?? resolveTaskType(taskPrompt, {
|
|
1536
|
+
fileTargets: parsePromptSpec(taskPrompt).fileConstraint.targets,
|
|
1537
|
+
changedFiles: evaluationContext ? [...evaluationContext.changedFiles] : void 0
|
|
1538
|
+
});
|
|
1539
|
+
const taskType = resolution.taskType;
|
|
1540
|
+
const limitedEmpiricalSignals = taskType !== "code";
|
|
1541
|
+
return {
|
|
1542
|
+
taskType,
|
|
1543
|
+
confidence: resolution.confidence,
|
|
1544
|
+
source: resolution.source,
|
|
1545
|
+
reasoning: resolution.reasoning,
|
|
1546
|
+
limitedEmpiricalSignals,
|
|
1547
|
+
guidance: buildTaskProfileGuidance({
|
|
1548
|
+
taskType,
|
|
1549
|
+
confidence: resolution.confidence,
|
|
1550
|
+
limitedEmpiricalSignals
|
|
1551
|
+
})
|
|
1552
|
+
};
|
|
1553
|
+
} catch (error) {
|
|
1554
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
1555
|
+
const taskType = "code";
|
|
1556
|
+
return {
|
|
1557
|
+
taskType,
|
|
1558
|
+
confidence: "low",
|
|
1559
|
+
source: "llm-fallback",
|
|
1560
|
+
reasoning: `Prompt profile fallback used due to resolver error: ${message}`,
|
|
1561
|
+
limitedEmpiricalSignals: false,
|
|
1562
|
+
guidance: buildTaskProfileGuidance({
|
|
1563
|
+
taskType,
|
|
1564
|
+
confidence: "low",
|
|
1565
|
+
limitedEmpiricalSignals: false
|
|
1566
|
+
})
|
|
1567
|
+
};
|
|
1568
|
+
}
|
|
1569
|
+
}
|
|
1570
|
+
/**
|
|
1571
|
+
* Map AIJudgeOutput to JudgeSynthesisResult.
|
|
1572
|
+
*/
|
|
1573
|
+
mapToSynthesisResult(output, variants, startTime, _preScoreGateContext) {
|
|
1574
|
+
const evaluations = output.variants.map((v) => {
|
|
1575
|
+
const matchingVariant = variants.find(
|
|
1576
|
+
(ov, idx) => ov.variant === v.id || v.id === `v${idx + 1}`
|
|
1577
|
+
);
|
|
1578
|
+
const variantName = matchingVariant?.variant ?? v.id;
|
|
1579
|
+
const readabilityScore = v.dimensions?.readability ?? v.quality.score;
|
|
1580
|
+
const maintainabilityScore = v.dimensions?.maintainability ?? v.quality.score;
|
|
1581
|
+
const innovationScore = v.dimensions?.innovation ?? 50;
|
|
1582
|
+
return {
|
|
1583
|
+
variant: variantName,
|
|
1584
|
+
declaredOverallScore: v.os,
|
|
1585
|
+
qualityScore: createPercentScore(v.quality.score, "quality"),
|
|
1586
|
+
strengths: [v.delivery.rationale, v.correctness.rationale, v.quality.rationale].filter(
|
|
1587
|
+
(r) => r && r.trim().length > 0
|
|
1588
|
+
),
|
|
1589
|
+
weaknesses: typeof v.limitation?.summary === "string" && v.limitation.summary.trim().length > 0 ? [v.limitation.summary] : [],
|
|
1590
|
+
primaryReason: `delivery=${v.delivery.score} correctness=${v.correctness.score} quality=${v.quality.score}`,
|
|
1591
|
+
codeQuality: {
|
|
1592
|
+
correctness: createPercentScore(v.correctness.score, "correctness"),
|
|
1593
|
+
readability: createPercentScore(readabilityScore, "readability"),
|
|
1594
|
+
maintainability: createPercentScore(maintainabilityScore, "maintainability"),
|
|
1595
|
+
performance: createPercentScore(70, "performance (not assessed)"),
|
|
1596
|
+
completeness: createPercentScore(v.delivery.score, "completeness"),
|
|
1597
|
+
innovation: createPercentScore(innovationScore, "innovation"),
|
|
1598
|
+
...v.dimensions?.security !== void 0 ? { security: createPercentScore(v.dimensions.security, "security") } : {}
|
|
1599
|
+
},
|
|
1600
|
+
bucketScores: {
|
|
1601
|
+
delivery: v.delivery.score,
|
|
1602
|
+
correctness: v.correctness.score,
|
|
1603
|
+
quality: v.quality.score
|
|
1604
|
+
}
|
|
1605
|
+
};
|
|
1606
|
+
});
|
|
1607
|
+
const deterministicWinner = selectDeterministicWinnerFromScores(
|
|
1608
|
+
evaluations.map((evaluation) => ({
|
|
1609
|
+
variant: evaluation.variant,
|
|
1610
|
+
eligible: true,
|
|
1611
|
+
declaredOverallScore: evaluation.declaredOverallScore,
|
|
1612
|
+
bucketScores: evaluation.bucketScores,
|
|
1613
|
+
codeQuality: evaluation.codeQuality,
|
|
1614
|
+
qualityScore: Number(evaluation.qualityScore ?? 0)
|
|
1615
|
+
}))
|
|
1616
|
+
);
|
|
1617
|
+
const variantNames = new Set(variants.map((v) => v.variant));
|
|
1618
|
+
const variantsList = variants.map((v) => ({ variant: v.variant }));
|
|
1619
|
+
const rawRanking = Array.isArray(output.summary?.ranking) ? output.summary.ranking : Array.isArray(output.ranking) ? output.ranking : void 0;
|
|
1620
|
+
const rankingValidation = validateRanking(rawRanking, variantNames, variantsList);
|
|
1621
|
+
const legacyWinnerRaw = typeof output.winner?.variant === "string" && output.winner.variant.trim().length > 0 ? output.winner.variant.trim() : typeof output.winner?.id === "string" && output.winner.id.trim().length > 0 ? output.winner.id.trim() : "";
|
|
1622
|
+
const legacyWinnerResolved = legacyWinnerRaw !== "" ? normalizeRankingIds([legacyWinnerRaw], variantsList)[0] ?? "" : "";
|
|
1623
|
+
const legacyWinnerValid = legacyWinnerResolved !== "" && variantNames.has(legacyWinnerResolved);
|
|
1624
|
+
const rankingLegacyDiverges = rankingValidation.valid && legacyWinnerValid && (rankingValidation.ranking[0] ?? null) !== legacyWinnerResolved;
|
|
1625
|
+
const explicitNoWinner = output.winner === null;
|
|
1626
|
+
if (variants.length === 1 && !explicitNoWinner) {
|
|
1627
|
+
const soleVariant = variants[0]?.variant;
|
|
1628
|
+
const sv01Raw = typeof output.summary?.confidence === "number" ? output.summary.confidence : 0;
|
|
1629
|
+
const sv01 = Math.max(0, Math.min(1, sv01Raw));
|
|
1630
|
+
const svRankingRationale = output.summary?.rankingRationale?.trim();
|
|
1631
|
+
const svReasoning = svRankingRationale && svRankingRationale.length > 0 ? svRankingRationale : `Single variant evaluation: ${soleVariant}.`;
|
|
1632
|
+
return {
|
|
1633
|
+
winner: soleVariant ?? null,
|
|
1634
|
+
evaluations,
|
|
1635
|
+
reasoning: svReasoning,
|
|
1636
|
+
winnerPrimaryReason: evaluations[0]?.primaryReason ?? "",
|
|
1637
|
+
recommendations: [],
|
|
1638
|
+
metadata: {
|
|
1639
|
+
model: this.model ?? "ai-judge",
|
|
1640
|
+
timestamp: /* @__PURE__ */ new Date(),
|
|
1641
|
+
evaluationTimeMs: Date.now() - startTime,
|
|
1642
|
+
confidence: createPercentScore(sv01 * 100, "single variant"),
|
|
1643
|
+
topVariants: [soleVariant],
|
|
1644
|
+
winnerSelectionMetadata: {
|
|
1645
|
+
rawWinner: soleVariant,
|
|
1646
|
+
finalWinner: soleVariant,
|
|
1647
|
+
selectionMode: "ranking",
|
|
1648
|
+
reason: "single_variant"
|
|
1649
|
+
},
|
|
1650
|
+
winnerSource: "ranking",
|
|
1651
|
+
audit: {
|
|
1652
|
+
deterministicWinner,
|
|
1653
|
+
legacyWinnerResolved: legacyWinnerResolved || null,
|
|
1654
|
+
legacyWinnerValid
|
|
1655
|
+
}
|
|
1656
|
+
}
|
|
1657
|
+
};
|
|
1658
|
+
}
|
|
1659
|
+
let winnerId = null;
|
|
1660
|
+
let winnerSource = "inconclusive";
|
|
1661
|
+
let validatedRanking = null;
|
|
1662
|
+
if (rankingValidation.valid && !rankingValidation.partial) {
|
|
1663
|
+
validatedRanking = rankingValidation.ranking;
|
|
1664
|
+
}
|
|
1665
|
+
if (explicitNoWinner) {
|
|
1666
|
+
winnerId = null;
|
|
1667
|
+
winnerSource = "inconclusive";
|
|
1668
|
+
} else if (legacyWinnerValid) {
|
|
1669
|
+
winnerId = legacyWinnerResolved;
|
|
1670
|
+
winnerSource = "llm-winner";
|
|
1671
|
+
} else if (validatedRanking) {
|
|
1672
|
+
winnerId = validatedRanking[0] ?? null;
|
|
1673
|
+
winnerSource = winnerId ? "ranking" : "inconclusive";
|
|
1674
|
+
}
|
|
1675
|
+
const rankingDivergesFromScores = winnerId != null && winnerId !== "" && winnerId !== deterministicWinner.winner;
|
|
1676
|
+
let rankingScoreInconsistency = false;
|
|
1677
|
+
if (validatedRanking && validatedRanking.length > 1) {
|
|
1678
|
+
const osMap = /* @__PURE__ */ new Map();
|
|
1679
|
+
for (const v of output.variants) {
|
|
1680
|
+
const vAny = v;
|
|
1681
|
+
if (typeof vAny.os === "number" && Number.isFinite(vAny.os)) {
|
|
1682
|
+
const match = variants.find((ov, idx) => ov.variant === v.id || v.id === `v${idx + 1}`);
|
|
1683
|
+
const key = match?.variant ?? v.id;
|
|
1684
|
+
osMap.set(key, vAny.os);
|
|
1685
|
+
}
|
|
1686
|
+
}
|
|
1687
|
+
if (osMap.size > 0) {
|
|
1688
|
+
const consistency = checkRankingScoreConsistency(validatedRanking, osMap);
|
|
1689
|
+
rankingScoreInconsistency = consistency.inconsistent;
|
|
1690
|
+
}
|
|
1691
|
+
}
|
|
1692
|
+
const winnerEvaluation = winnerId ? evaluations.find((evaluation) => evaluation.variant === winnerId) : void 0;
|
|
1693
|
+
let selfConsistencyWarning;
|
|
1694
|
+
if (winnerId && winnerSource !== "inconclusive") {
|
|
1695
|
+
selfConsistencyWarning = computeSelfConsistencyWarning(winnerId, deterministicWinner);
|
|
1696
|
+
}
|
|
1697
|
+
const summaryReason = typeof output.summary?.rankingRationale === "string" && output.summary.rankingRationale.trim() ? output.summary.rankingRationale.trim() : null;
|
|
1698
|
+
const fallbackReason = winnerEvaluation?.primaryReason ?? "Winner derived from validated ranking.";
|
|
1699
|
+
const reasoning = summaryReason ?? fallbackReason;
|
|
1700
|
+
const confidence01Raw = typeof output.summary?.confidence === "number" ? output.summary.confidence : 0;
|
|
1701
|
+
const confidence01 = Math.max(0, Math.min(1, confidence01Raw));
|
|
1702
|
+
const topVariants = (validatedRanking ?? deterministicWinner.scored.map((entry) => entry.variant)).slice(0, 3);
|
|
1703
|
+
return {
|
|
1704
|
+
winner: winnerId,
|
|
1705
|
+
evaluations,
|
|
1706
|
+
reasoning,
|
|
1707
|
+
recommendations: [],
|
|
1708
|
+
metadata: {
|
|
1709
|
+
model: this.model ?? "ai-judge",
|
|
1710
|
+
timestamp: /* @__PURE__ */ new Date(),
|
|
1711
|
+
evaluationTimeMs: Date.now() - startTime,
|
|
1712
|
+
confidence: createPercentScore(confidence01 * 100, "ai judge confidence"),
|
|
1713
|
+
topVariants,
|
|
1714
|
+
winnerSelectionMetadata: {
|
|
1715
|
+
rawWinner: legacyWinnerRaw || null,
|
|
1716
|
+
rawRanking: Array.isArray(rawRanking) ? rawRanking : null,
|
|
1717
|
+
validatedRanking,
|
|
1718
|
+
rankingLegacyDiverges,
|
|
1719
|
+
finalWinner: winnerId,
|
|
1720
|
+
selectionMode: winnerSource === "llm-winner" ? "llm-winner" : winnerSource === "ranking" ? "ranking" : "ranking-inconclusive",
|
|
1721
|
+
reason: explicitNoWinner ? "explicit_no_winner" : winnerSource === "llm-winner" ? "explicit_winner" : winnerSource === "ranking" ? "ranking_first" : "llm_no_declaration",
|
|
1722
|
+
margin: deterministicWinner.margin,
|
|
1723
|
+
thresholds: deterministicWinner.thresholds,
|
|
1724
|
+
scored: deterministicWinner.scored.map((entry) => ({
|
|
1725
|
+
variant: entry.variant,
|
|
1726
|
+
score: Number(entry.compositeScore),
|
|
1727
|
+
source: entry.scoreSource
|
|
1728
|
+
}))
|
|
1729
|
+
},
|
|
1730
|
+
audit: {
|
|
1731
|
+
judgeType: "ai-judge",
|
|
1732
|
+
deterministicWinner,
|
|
1733
|
+
winnerSource,
|
|
1734
|
+
rankingValidation: rankingValidation.valid ? rankingValidation.partial ? "partial_rejected" : "valid" : rankingValidation.reason,
|
|
1735
|
+
rankingDivergesFromScores,
|
|
1736
|
+
rankingScoreInconsistency,
|
|
1737
|
+
rankingLegacyDiverges,
|
|
1738
|
+
validatedRanking,
|
|
1739
|
+
rawRanking: Array.isArray(rawRanking) ? rawRanking : null,
|
|
1740
|
+
llmWinnerAuthority: {
|
|
1741
|
+
llmDeclaredVariantRaw: legacyWinnerRaw || null,
|
|
1742
|
+
llmDeclaredVariant: legacyWinnerResolved || null,
|
|
1743
|
+
llmDeclaredNoWinner: explicitNoWinner,
|
|
1744
|
+
llmWinnerValid: explicitNoWinner || legacyWinnerValid,
|
|
1745
|
+
used: explicitNoWinner ? "explicit-no-winner" : winnerSource
|
|
1746
|
+
},
|
|
1747
|
+
...selfConsistencyWarning ? { selfConsistencyWarning } : {}
|
|
1748
|
+
}
|
|
1749
|
+
},
|
|
1750
|
+
winnerPrimaryReason: winnerEvaluation?.primaryReason ?? void 0
|
|
1751
|
+
};
|
|
1752
|
+
}
|
|
1753
|
+
/**
|
|
1754
|
+
* Factual-consistency audit (telemetry-only, NO mutation).
|
|
1755
|
+
*
|
|
1756
|
+
* After LLM output is parsed and validated, checks if AI rationale claims
|
|
1757
|
+
* tests passed when gate data shows 'failed' or 'error'. Records warnings
|
|
1758
|
+
* to synthesis.metadata.audit but does NOT mutate confidence, scores, or winner.
|
|
1759
|
+
*/
|
|
1760
|
+
validateFactualConsistency(output, variants, preScoreGateContext, synthesis) {
|
|
1761
|
+
if (!preScoreGateContext.isCodeTask || preScoreGateContext.gateResultsByVariant.size === 0) {
|
|
1762
|
+
return;
|
|
1763
|
+
}
|
|
1764
|
+
const warnings = [];
|
|
1765
|
+
const testPassClaimPattern = /\b(?:all tests? pass(?:ed|ing)?|tests? (?:are )?passing|0 fail(?:ure)?s?|no (?:test )?fail(?:ure)?s?|clean test (?:run|pass))\b/i;
|
|
1766
|
+
for (const variantOutput of output.variants) {
|
|
1767
|
+
const rationales = [
|
|
1768
|
+
variantOutput.delivery?.rationale,
|
|
1769
|
+
variantOutput.correctness?.rationale,
|
|
1770
|
+
variantOutput.quality?.rationale
|
|
1771
|
+
].filter((r) => typeof r === "string" && r.length > 0).join(" ");
|
|
1772
|
+
if (!testPassClaimPattern.test(rationales)) continue;
|
|
1773
|
+
const resolvedVariant = variants.find(
|
|
1774
|
+
(variant, index) => variant.variant === variantOutput.id || variantOutput.id === `v${index + 1}`
|
|
1775
|
+
);
|
|
1776
|
+
const gateVariantName = resolvedVariant?.variant ?? variantOutput.id;
|
|
1777
|
+
const gateResults = preScoreGateContext.gateResultsByVariant.get(gateVariantName);
|
|
1778
|
+
if (!gateResults) continue;
|
|
1779
|
+
const testGate = gateResults.testGate;
|
|
1780
|
+
if (testGate && (testGate.testStatus === "failed" || testGate.testStatus === "error")) {
|
|
1781
|
+
warnings.push(
|
|
1782
|
+
`Variant ${variantOutput.id} (${gateVariantName}): AI rationale claims tests passed but testGate.testStatus=${testGate.testStatus} (exitCode=${testGate.exitCode})`
|
|
1783
|
+
);
|
|
1784
|
+
}
|
|
1785
|
+
}
|
|
1786
|
+
const existingAudit = synthesis.metadata.audit ?? {};
|
|
1787
|
+
synthesis.metadata.audit = {
|
|
1788
|
+
...existingAudit,
|
|
1789
|
+
factualConsistencyPassed: warnings.length === 0,
|
|
1790
|
+
factualConsistencyWarnings: warnings.length > 0 ? warnings : void 0
|
|
1791
|
+
};
|
|
1792
|
+
}
|
|
1793
|
+
/**
|
|
1794
|
+
* Persist a failed AI response to disk for post-mortem analysis.
|
|
1795
|
+
* Never throws — failures here must not block the judge error path.
|
|
1796
|
+
*/
|
|
1797
|
+
persistFailedResponse(response, taskId, label) {
|
|
1798
|
+
persistJudgeFailureArtifact({ response, taskId, label });
|
|
1799
|
+
}
|
|
1800
|
+
repairMissingOverallScores(parsed) {
|
|
1801
|
+
if (!parsed || typeof parsed !== "object") return parsed;
|
|
1802
|
+
const root = parsed;
|
|
1803
|
+
if (!Array.isArray(root.variants) || root.variants.length <= 1) return parsed;
|
|
1804
|
+
const repairedVariants = [];
|
|
1805
|
+
let changed = false;
|
|
1806
|
+
for (const variant of root.variants) {
|
|
1807
|
+
if (!variant || typeof variant !== "object") return parsed;
|
|
1808
|
+
const entry = variant;
|
|
1809
|
+
if (entry.os !== void 0) {
|
|
1810
|
+
repairedVariants.push(entry);
|
|
1811
|
+
continue;
|
|
1812
|
+
}
|
|
1813
|
+
const delivery = this.readBucketScore(entry.delivery);
|
|
1814
|
+
const correctness = this.readBucketScore(entry.correctness);
|
|
1815
|
+
const quality = this.readBucketScore(entry.quality);
|
|
1816
|
+
if (delivery === void 0 || correctness === void 0 || quality === void 0) {
|
|
1817
|
+
return parsed;
|
|
1818
|
+
}
|
|
1819
|
+
changed = true;
|
|
1820
|
+
repairedVariants.push({
|
|
1821
|
+
...entry,
|
|
1822
|
+
os: Math.round(computeFinalScore({ delivery, correctness, quality }) * 10) / 10
|
|
1823
|
+
});
|
|
1824
|
+
}
|
|
1825
|
+
if (changed) {
|
|
1826
|
+
const repairedCount = repairedVariants.filter((v, i) => {
|
|
1827
|
+
const original = root.variants[i];
|
|
1828
|
+
return original?.os === void 0 && v.os !== void 0;
|
|
1829
|
+
}).length;
|
|
1830
|
+
console.warn(
|
|
1831
|
+
`[ai-judge] repairMissingOverallScores: synthesized os from bucket scores for ${repairedCount} variant(s)`
|
|
1832
|
+
);
|
|
1833
|
+
return { ...root, variants: repairedVariants };
|
|
1834
|
+
}
|
|
1835
|
+
return parsed;
|
|
1836
|
+
}
|
|
1837
|
+
readBucketScore(bucket) {
|
|
1838
|
+
if (!bucket || typeof bucket !== "object") return void 0;
|
|
1839
|
+
const score = bucket.score;
|
|
1840
|
+
return typeof score === "number" && Number.isFinite(score) && score >= 0 && score <= 100 ? score : void 0;
|
|
1841
|
+
}
|
|
1842
|
+
classifyParseFailureReason(response) {
|
|
1843
|
+
const trimmed = response.trim();
|
|
1844
|
+
if (trimmed.length === 0) return "empty_output";
|
|
1845
|
+
const hasJsonLikeMarkers = /[{[]/.test(trimmed);
|
|
1846
|
+
return hasJsonLikeMarkers ? "malformed_json" : "no_json_payload";
|
|
1847
|
+
}
|
|
1848
|
+
createResponseSnippet(response) {
|
|
1849
|
+
const normalized = response.replace(/\s+/g, " ").trim();
|
|
1850
|
+
if (normalized.length === 0) return "<empty output>";
|
|
1851
|
+
const snippet = normalized.slice(0, 300);
|
|
1852
|
+
return normalized.length > 300 ? `${snippet}...` : snippet;
|
|
1853
|
+
}
|
|
1854
|
+
describeValidationFailure(parsed) {
|
|
1855
|
+
if (!parsed || typeof parsed !== "object") {
|
|
1856
|
+
return `top-level payload is ${parsed === null ? "null" : typeof parsed}`;
|
|
1857
|
+
}
|
|
1858
|
+
const root = parsed;
|
|
1859
|
+
const keys = Object.keys(root);
|
|
1860
|
+
if (!Array.isArray(root.variants)) {
|
|
1861
|
+
return `missing variants array (keys=${keys.join(",") || "none"})`;
|
|
1862
|
+
}
|
|
1863
|
+
if (root.variants.length === 0) {
|
|
1864
|
+
return "variants array is empty";
|
|
1865
|
+
}
|
|
1866
|
+
if (root.variants.length > 1) {
|
|
1867
|
+
const missingOverallScoreIndexes = root.variants.map(
|
|
1868
|
+
(variant, index) => variant && typeof variant === "object" && variant.os === void 0 ? index : -1
|
|
1869
|
+
).filter((index) => index >= 0);
|
|
1870
|
+
if (missingOverallScoreIndexes.length > 0) {
|
|
1871
|
+
return `multi-variant output missing os for variant indexes ${missingOverallScoreIndexes.join(",")}`;
|
|
1872
|
+
}
|
|
1873
|
+
}
|
|
1874
|
+
return `invalid variants/summary structure (keys=${keys.join(",") || "none"})`;
|
|
1875
|
+
}
|
|
1876
|
+
/**
|
|
1877
|
+
* Create an empty result for edge cases.
|
|
1878
|
+
*/
|
|
1879
|
+
createEmptyResult(reason, startTime) {
|
|
1880
|
+
return {
|
|
1881
|
+
winner: null,
|
|
1882
|
+
evaluations: [],
|
|
1883
|
+
reasoning: reason,
|
|
1884
|
+
recommendations: ["Provide at least one variant to evaluate"],
|
|
1885
|
+
metadata: {
|
|
1886
|
+
model: this.model ?? "ai-judge",
|
|
1887
|
+
timestamp: /* @__PURE__ */ new Date(),
|
|
1888
|
+
evaluationTimeMs: Date.now() - startTime,
|
|
1889
|
+
confidence: createPercentScore(0, "no variants"),
|
|
1890
|
+
topVariants: []
|
|
1891
|
+
}
|
|
1892
|
+
};
|
|
1893
|
+
}
|
|
1894
|
+
};
|
|
1895
|
+
function createAIJudge(options) {
|
|
1896
|
+
return new AIJudge(options);
|
|
1897
|
+
}
|
|
1898
|
+
|
|
1899
|
+
export {
|
|
1900
|
+
buildTaskProfileGuidance,
|
|
1901
|
+
parseJudgeOutput,
|
|
1902
|
+
AIJudge,
|
|
1903
|
+
createAIJudge
|
|
1904
|
+
};
|
|
1905
|
+
//# sourceMappingURL=chunk-CKP2Q3TP.js.map
|