@poetic-ai/poetic 1.42.1-bootstrap.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/SECURITY.md +47 -0
- package/.nvmrc +1 -0
- package/.poetic/README.md +37 -0
- package/.poetic/providers/catalog.json +12051 -0
- package/.poetic/providers/pricing.json +1272 -0
- package/.poetic/providers/registry.json +3369 -0
- package/CHANGELOG.md +552 -0
- package/CODE_OF_CONDUCT.md +40 -0
- package/CONTRIBUTING.md +23 -0
- package/INSTALL.md +454 -0
- package/LICENSE +21 -0
- package/README.md +474 -0
- package/dist/BasicOptimizationCompetitionRunner-5H5TBYO4.js +153 -0
- package/dist/ExecutionTracker-UVONC4C6.js +16 -0
- package/dist/OptimizationConfig-7DFY2TST.js +17 -0
- package/dist/OptimizationEngine-MXTSOSST.js +1597 -0
- package/dist/PromptStore-U6ZD3ZMQ.js +18 -0
- package/dist/actions-M3AESHFF.js +322 -0
- package/dist/agent-ingest-HO77TH26.js +53 -0
- package/dist/aggregator-DNCBINFO.js +12 -0
- package/dist/ai-judge-LDWHURKK.js +156 -0
- package/dist/allowlist-grounding-RO6HKUFI.js +16 -0
- package/dist/allowlist-utils-23V7TJ3G.js +28 -0
- package/dist/anchored-turn-service-GFYU6GCQ.js +489 -0
- package/dist/api-transport-FS3CQRMN.js +1087 -0
- package/dist/apply-completion-mode-JBT2CTQO.js +89 -0
- package/dist/artifact-migrator-23L45ISD.js +267 -0
- package/dist/ask-UYVZY7JH.js +135 -0
- package/dist/auth-CSQP5RHJ.js +67 -0
- package/dist/auth-E6XUNZ5M.js +68 -0
- package/dist/auth-KQLJPZ2E.js +134 -0
- package/dist/auth-OC4E6633.js +71 -0
- package/dist/auth-R3NQ2ZF5.js +78 -0
- package/dist/auth-S4CTX4GQ.js +44 -0
- package/dist/auth-YLH3S7EY.js +69 -0
- package/dist/auth-liveness-TDDEZY3T.js +46 -0
- package/dist/auto-optimizer-HA3T23SF.js +80 -0
- package/dist/autoloop-cleanup-JOGKTLWA.js +109 -0
- package/dist/autoloop-evidence-gates-ACB5YW5G.js +36 -0
- package/dist/autoloop-gate-policy-XZA5EQFN.js +175 -0
- package/dist/autoloop-helpers-EFJLBQUP.js +22 -0
- package/dist/autoloop-ledger-W7QO62PX.js +96 -0
- package/dist/autoloop-ledger-subscriber-Z5RW5EIE.js +162 -0
- package/dist/autoloop-run-defaults-FERGRZGP.js +19 -0
- package/dist/autoloop-run-lock-CT2NWSV3.js +197 -0
- package/dist/autoloop-success-OPR74YEX.js +85 -0
- package/dist/autoloop-termination-summary-RN75F422.js +58 -0
- package/dist/autoloop-trajectory-K5I45UBR.js +282 -0
- package/dist/autoloop-wall-clock-T7ZVIQOC.js +12 -0
- package/dist/autonomous-loop-controller-4QJMOIIN.js +4120 -0
- package/dist/backend-P3DAYRP7.js +29 -0
- package/dist/background-executor-CBN552V4.js +257 -0
- package/dist/backlog-execution-intent-U4XYPTV6.js +15 -0
- package/dist/backup-active-store-5WXG6SXT.js +17 -0
- package/dist/backup-telemetry-db-XGLXAA3N.js +254 -0
- package/dist/basic-competition-master-WXZHUT42.js +344 -0
- package/dist/branch-archive-manager-RP6FP3ZC.js +254 -0
- package/dist/branch-cleanup-manager-RSHAI7YR.js +20 -0
- package/dist/build-gate-YFASRB5Z.js +71 -0
- package/dist/change-summary-WY7J5TOX.js +33 -0
- package/dist/chunk-25QL3M3L.js +899 -0
- package/dist/chunk-2ABC6SEC.js +2626 -0
- package/dist/chunk-2AICM4P3.js +1763 -0
- package/dist/chunk-2AQUWJPW.js +225 -0
- package/dist/chunk-2DY5KNVM.js +276 -0
- package/dist/chunk-2QJ3H3L7.js +42 -0
- package/dist/chunk-2UH6VQRZ.js +328 -0
- package/dist/chunk-2VSQMRCB.js +18 -0
- package/dist/chunk-2XQXYBM2.js +130 -0
- package/dist/chunk-2YFUIER7.js +426 -0
- package/dist/chunk-2YXZCZTX.js +1160 -0
- package/dist/chunk-32VLDFRT.js +797 -0
- package/dist/chunk-34CNP2HB.js +148 -0
- package/dist/chunk-35P3W3JX.js +456 -0
- package/dist/chunk-36PEBZAF.js +18 -0
- package/dist/chunk-3733QTI7.js +6818 -0
- package/dist/chunk-3A5PDH5L.js +12513 -0
- package/dist/chunk-3AOKYKA7.js +22 -0
- package/dist/chunk-3BCUNHKF.js +475 -0
- package/dist/chunk-3C26DPFK.js +566 -0
- package/dist/chunk-3FLXLLB7.js +26 -0
- package/dist/chunk-3FOZ2VHV.js +100 -0
- package/dist/chunk-3G6FZCRF.js +106 -0
- package/dist/chunk-3JWNRJSB.js +736 -0
- package/dist/chunk-3LUMR3D3.js +156 -0
- package/dist/chunk-3M57DADF.js +44 -0
- package/dist/chunk-3S62LLWJ.js +543 -0
- package/dist/chunk-3VABAKT2.js +1784 -0
- package/dist/chunk-44N5WZZH.js +45 -0
- package/dist/chunk-45FPSA3B.js +125 -0
- package/dist/chunk-4BIPZBT7.js +791 -0
- package/dist/chunk-4BJ3O4RQ.js +929 -0
- package/dist/chunk-4BTDUT2S.js +42 -0
- package/dist/chunk-4CSS7WVI.js +143 -0
- package/dist/chunk-4CUY7AFY.js +247 -0
- package/dist/chunk-4HN6DF7V.js +1027 -0
- package/dist/chunk-4HXKHDNH.js +1064 -0
- package/dist/chunk-4KKI477Q.js +44 -0
- package/dist/chunk-4NA7FSTV.js +595 -0
- package/dist/chunk-4NXPXE62.js +84 -0
- package/dist/chunk-4SALLK63.js +9831 -0
- package/dist/chunk-4SHKPCGK.js +110 -0
- package/dist/chunk-4VJITV4P.js +445 -0
- package/dist/chunk-4W2NRXQL.js +1205 -0
- package/dist/chunk-4XICK3HN.js +784 -0
- package/dist/chunk-55EJVV3C.js +385 -0
- package/dist/chunk-5DLKYTQX.js +4003 -0
- package/dist/chunk-5EERLVZB.js +139 -0
- package/dist/chunk-5HDP7XZG.js +91 -0
- package/dist/chunk-5LAMRH4X.js +969 -0
- package/dist/chunk-5REDMNLZ.js +310 -0
- package/dist/chunk-5UCJARII.js +1984 -0
- package/dist/chunk-5WCQRDRB.js +588 -0
- package/dist/chunk-5WRLK5KW.js +700 -0
- package/dist/chunk-66AGFTBQ.js +131 -0
- package/dist/chunk-66CIO2SV.js +48 -0
- package/dist/chunk-6BM72ZMG.js +31 -0
- package/dist/chunk-6BTFXYRO.js +276 -0
- package/dist/chunk-6EDAJH2I.js +68 -0
- package/dist/chunk-6EQKFDBH.js +457 -0
- package/dist/chunk-6KMJKHIQ.js +2193 -0
- package/dist/chunk-6MFRHTRI.js +115 -0
- package/dist/chunk-6QMUKUHH.js +632 -0
- package/dist/chunk-6TZJRKNW.js +321 -0
- package/dist/chunk-6V7HGQFZ.js +20 -0
- package/dist/chunk-6Y5TWI7H.js +146 -0
- package/dist/chunk-6Y5U7UFY.js +91 -0
- package/dist/chunk-6ZLAK3XG.js +14 -0
- package/dist/chunk-72XCRQED.js +34 -0
- package/dist/chunk-72YTZMAM.js +589 -0
- package/dist/chunk-73NCTWKL.js +188 -0
- package/dist/chunk-77I6G4CO.js +1 -0
- package/dist/chunk-7D7QT3BE.js +688 -0
- package/dist/chunk-7DNSNKJV.js +78 -0
- package/dist/chunk-7FMJVYBS.js +307 -0
- package/dist/chunk-7GDRQBQC.js +419 -0
- package/dist/chunk-7IL4A2PP.js +948 -0
- package/dist/chunk-7NPSXSHO.js +975 -0
- package/dist/chunk-7PS6INZP.js +302 -0
- package/dist/chunk-7QCZSLBH.js +1 -0
- package/dist/chunk-7R2WLDOS.js +62 -0
- package/dist/chunk-7XV3QXLT.js +878 -0
- package/dist/chunk-A2WKSVX7.js +43 -0
- package/dist/chunk-A3YGID55.js +133 -0
- package/dist/chunk-A5MRWHMI.js +1604 -0
- package/dist/chunk-A722DEVA.js +14 -0
- package/dist/chunk-AG2Z2SUZ.js +554 -0
- package/dist/chunk-AHWAZ2MU.js +137 -0
- package/dist/chunk-AIAQ3HS3.js +119 -0
- package/dist/chunk-ALAB3PBR.js +2206 -0
- package/dist/chunk-APPNGGN7.js +365 -0
- package/dist/chunk-APV6MK5E.js +137 -0
- package/dist/chunk-AQNPGRWS.js +395 -0
- package/dist/chunk-ASEP3M2W.js +968 -0
- package/dist/chunk-AT4TNPWW.js +2931 -0
- package/dist/chunk-ATVKCSGT.js +250 -0
- package/dist/chunk-AX6GWE7V.js +47 -0
- package/dist/chunk-AZQILSNQ.js +1376 -0
- package/dist/chunk-B76KHLX6.js +13 -0
- package/dist/chunk-BAV2HJXS.js +194 -0
- package/dist/chunk-BBHRN366.js +89 -0
- package/dist/chunk-BCZAENOH.js +448 -0
- package/dist/chunk-BIGSACCY.js +19527 -0
- package/dist/chunk-BPYMCIVD.js +9243 -0
- package/dist/chunk-BSE4R6XD.js +322 -0
- package/dist/chunk-BVPWRDCO.js +2321 -0
- package/dist/chunk-BXI4NXL3.js +507 -0
- package/dist/chunk-C3L6YQ7P.js +147 -0
- package/dist/chunk-C57KJOUQ.js +121 -0
- package/dist/chunk-CCV7BPSY.js +118 -0
- package/dist/chunk-CD7ISXA4.js +826 -0
- package/dist/chunk-CE5OZBYY.js +21 -0
- package/dist/chunk-CFBIG37O.js +154 -0
- package/dist/chunk-CHQV73H5.js +408 -0
- package/dist/chunk-CJC6GZ46.js +457 -0
- package/dist/chunk-CJHRFSNY.js +924 -0
- package/dist/chunk-CKP2Q3TP.js +1905 -0
- package/dist/chunk-CLKSETWE.js +10 -0
- package/dist/chunk-CQM3A35X.js +844 -0
- package/dist/chunk-CSLAS5GW.js +4007 -0
- package/dist/chunk-CTLUBNCW.js +622 -0
- package/dist/chunk-CVMVZWG4.js +3544 -0
- package/dist/chunk-CYB6QDCT.js +263 -0
- package/dist/chunk-D5EP5D2W.js +26 -0
- package/dist/chunk-DC4ZYMXJ.js +115 -0
- package/dist/chunk-DCAVAABI.js +348 -0
- package/dist/chunk-DCYF7EKG.js +26 -0
- package/dist/chunk-DF5SLDE4.js +966 -0
- package/dist/chunk-DN7UMLF7.js +5678 -0
- package/dist/chunk-DSFM56OA.js +1203 -0
- package/dist/chunk-DTY4APYV.js +218 -0
- package/dist/chunk-E3NTPHEO.js +194 -0
- package/dist/chunk-E74LUIID.js +302 -0
- package/dist/chunk-EJ42BNGY.js +2904 -0
- package/dist/chunk-EJ4ZAWLV.js +305 -0
- package/dist/chunk-EJGCZGUF.js +1483 -0
- package/dist/chunk-EK2U3YEG.js +486 -0
- package/dist/chunk-ELMFYNV7.js +109 -0
- package/dist/chunk-ENRGKNR5.js +25 -0
- package/dist/chunk-EQGIDAUF.js +62 -0
- package/dist/chunk-EX3OM35G.js +20414 -0
- package/dist/chunk-F5JO7HAS.js +94 -0
- package/dist/chunk-F7ER6ANO.js +174 -0
- package/dist/chunk-FATUVIIN.js +374 -0
- package/dist/chunk-FC23CTXV.js +322 -0
- package/dist/chunk-FCNAS3W2.js +1033 -0
- package/dist/chunk-FD4ERYEG.js +22 -0
- package/dist/chunk-FDVZLG6Z.js +3104 -0
- package/dist/chunk-FEPIFY7D.js +511 -0
- package/dist/chunk-FEQ4HXL7.js +412 -0
- package/dist/chunk-FFH5HHJU.js +5351 -0
- package/dist/chunk-FH7WOJAE.js +79 -0
- package/dist/chunk-FKAUPHMR.js +96 -0
- package/dist/chunk-FRG57FPM.js +240 -0
- package/dist/chunk-FUVP64ER.js +31 -0
- package/dist/chunk-G2PTLLEL.js +284 -0
- package/dist/chunk-G35UYQ6Z.js +69 -0
- package/dist/chunk-G36W2T2S.js +1143 -0
- package/dist/chunk-G4THLFV3.js +148 -0
- package/dist/chunk-GAFJEYLK.js +29 -0
- package/dist/chunk-GCCAPO3S.js +118 -0
- package/dist/chunk-GG6KJYW6.js +85 -0
- package/dist/chunk-GHHFDU2U.js +55 -0
- package/dist/chunk-GHTZSQB7.js +57 -0
- package/dist/chunk-GJTWJNLV.js +1374 -0
- package/dist/chunk-GKN3HMN4.js +2809 -0
- package/dist/chunk-GNH3QGTL.js +1342 -0
- package/dist/chunk-GOAQ2I7Y.js +7000 -0
- package/dist/chunk-GSGD5E4C.js +30 -0
- package/dist/chunk-GXWAJQOJ.js +5011 -0
- package/dist/chunk-H2AYQIHW.js +224 -0
- package/dist/chunk-H6G7QRMQ.js +35 -0
- package/dist/chunk-H7JAGHZT.js +59 -0
- package/dist/chunk-HAC6EHZZ.js +79 -0
- package/dist/chunk-HE2KH6EQ.js +161 -0
- package/dist/chunk-HIHENQDX.js +66 -0
- package/dist/chunk-HIO33P3G.js +19 -0
- package/dist/chunk-HK4AQZSR.js +438 -0
- package/dist/chunk-HMVQIPH3.js +547 -0
- package/dist/chunk-HOYBYAOA.js +257 -0
- package/dist/chunk-HQLGXODD.js +123 -0
- package/dist/chunk-HQXYYUNY.js +1231 -0
- package/dist/chunk-HY3ESWCA.js +708 -0
- package/dist/chunk-I46EG2XQ.js +65 -0
- package/dist/chunk-I4NPDSCB.js +279 -0
- package/dist/chunk-I5WRT7ZR.js +514 -0
- package/dist/chunk-I6BL3PDH.js +1024 -0
- package/dist/chunk-IATQ7NBI.js +679 -0
- package/dist/chunk-IC7I6YIJ.js +301 -0
- package/dist/chunk-IDCSOZVN.js +78 -0
- package/dist/chunk-IEXU4RR6.js +5742 -0
- package/dist/chunk-IFCBYECK.js +112 -0
- package/dist/chunk-IFZCPEY2.js +66 -0
- package/dist/chunk-IKFP2KUR.js +48 -0
- package/dist/chunk-IP7BYVUV.js +338 -0
- package/dist/chunk-IRJ2I6KH.js +166 -0
- package/dist/chunk-IUXNGKA6.js +152 -0
- package/dist/chunk-IZYIQQ7X.js +53 -0
- package/dist/chunk-IZZK3H6I.js +2175 -0
- package/dist/chunk-J3J4P7RW.js +353 -0
- package/dist/chunk-J62KPGII.js +464 -0
- package/dist/chunk-J6P3FWWW.js +130 -0
- package/dist/chunk-JD6SIZED.js +712 -0
- package/dist/chunk-JEV6UI7H.js +537 -0
- package/dist/chunk-JKBHJQMT.js +434 -0
- package/dist/chunk-JLS65NUQ.js +14 -0
- package/dist/chunk-JOSYEJKY.js +141 -0
- package/dist/chunk-JPKSXKNB.js +94 -0
- package/dist/chunk-JQD5VYKR.js +483 -0
- package/dist/chunk-JSOWEFEQ.js +351 -0
- package/dist/chunk-JSWMRQYG.js +28 -0
- package/dist/chunk-K4XK7SGB.js +6783 -0
- package/dist/chunk-K54Y7MLK.js +160 -0
- package/dist/chunk-KAT4RXUT.js +521 -0
- package/dist/chunk-KDMKKIRK.js +103 -0
- package/dist/chunk-KM3F7SYE.js +702 -0
- package/dist/chunk-KNDYMVWN.js +102 -0
- package/dist/chunk-KNT72VR6.js +132 -0
- package/dist/chunk-KQEBSK3T.js +274 -0
- package/dist/chunk-KT6VHSPF.js +84 -0
- package/dist/chunk-KTMW65UX.js +14 -0
- package/dist/chunk-L4427KOE.js +26 -0
- package/dist/chunk-L6OM4A22.js +189 -0
- package/dist/chunk-L7E6XCRA.js +1042 -0
- package/dist/chunk-L7JZLXQL.js +173 -0
- package/dist/chunk-LAMKBTXX.js +40 -0
- package/dist/chunk-LDCF2MKN.js +132 -0
- package/dist/chunk-LEIIWPWQ.js +78 -0
- package/dist/chunk-LFTQVYBB.js +31 -0
- package/dist/chunk-LGHII5ET.js +23 -0
- package/dist/chunk-LH34OGSB.js +265 -0
- package/dist/chunk-LO3X5NLS.js +326 -0
- package/dist/chunk-LPITVM7M.js +339 -0
- package/dist/chunk-LR5IJIGV.js +810 -0
- package/dist/chunk-LUVKFFWR.js +315 -0
- package/dist/chunk-LWCBNGH6.js +35 -0
- package/dist/chunk-LWTN3WRV.js +260 -0
- package/dist/chunk-LZBBXYWF.js +1004 -0
- package/dist/chunk-M34PS4NE.js +2213 -0
- package/dist/chunk-M3HQYKQX.js +308 -0
- package/dist/chunk-MCPQNJMK.js +482 -0
- package/dist/chunk-MF5HP6XV.js +221 -0
- package/dist/chunk-MFADRRTR.js +112 -0
- package/dist/chunk-MN6PL4AN.js +120 -0
- package/dist/chunk-MQ4WA34C.js +206 -0
- package/dist/chunk-MTVZITQ2.js +185 -0
- package/dist/chunk-MWHF5V7U.js +223 -0
- package/dist/chunk-MWHLBPNU.js +211 -0
- package/dist/chunk-MWK5UZQ3.js +1068 -0
- package/dist/chunk-MYLH2S4G.js +64 -0
- package/dist/chunk-MZ5WYKNA.js +28 -0
- package/dist/chunk-N5DX4JAK.js +12 -0
- package/dist/chunk-NEDCMN7E.js +415 -0
- package/dist/chunk-NEQUQHOX.js +223 -0
- package/dist/chunk-NGK6PEOG.js +25 -0
- package/dist/chunk-NM6PMYY4.js +768 -0
- package/dist/chunk-NOZGR2QO.js +2373 -0
- package/dist/chunk-NP2V3L7K.js +226 -0
- package/dist/chunk-NPZJMUZ4.js +194 -0
- package/dist/chunk-NRLY536M.js +260 -0
- package/dist/chunk-NS33V3IM.js +51 -0
- package/dist/chunk-NWOYOA6H.js +378 -0
- package/dist/chunk-NWRUSEXG.js +322 -0
- package/dist/chunk-O6V4PNLO.js +63 -0
- package/dist/chunk-OGMSYL6J.js +227 -0
- package/dist/chunk-OIS7J26S.js +319 -0
- package/dist/chunk-OKBTRSZI.js +205 -0
- package/dist/chunk-OMX5VL43.js +79 -0
- package/dist/chunk-ON3G73BU.js +264 -0
- package/dist/chunk-OPYFYTWQ.js +220 -0
- package/dist/chunk-OQ6K46CK.js +1734 -0
- package/dist/chunk-OSDYVKHW.js +2440 -0
- package/dist/chunk-OUM5S64R.js +3429 -0
- package/dist/chunk-OWB7JXUS.js +751 -0
- package/dist/chunk-OYZYO7TV.js +47 -0
- package/dist/chunk-P47VN5F4.js +282 -0
- package/dist/chunk-P6VQROCO.js +594 -0
- package/dist/chunk-P7CQGPLS.js +30 -0
- package/dist/chunk-PA6BCSOX.js +2872 -0
- package/dist/chunk-PGSPX4SU.js +15 -0
- package/dist/chunk-PIKLF7BM.js +432 -0
- package/dist/chunk-PJWJ3SAV.js +22 -0
- package/dist/chunk-PKIFMV72.js +100 -0
- package/dist/chunk-PTH5E5XO.js +195 -0
- package/dist/chunk-PYITM4N2.js +79 -0
- package/dist/chunk-PZ5AY32C.js +10 -0
- package/dist/chunk-Q3RO7N35.js +45 -0
- package/dist/chunk-Q3WIC6GQ.js +2670 -0
- package/dist/chunk-QGJDMEOV.js +34 -0
- package/dist/chunk-QJYIHVXT.js +100 -0
- package/dist/chunk-QNOB37UH.js +58 -0
- package/dist/chunk-QOCOVLUH.js +462 -0
- package/dist/chunk-QTCJ5CC5.js +172 -0
- package/dist/chunk-QU6JQJMH.js +3063 -0
- package/dist/chunk-QVFV5IP2.js +232 -0
- package/dist/chunk-QVZMFDYG.js +29 -0
- package/dist/chunk-R2ZYWNW3.js +1581 -0
- package/dist/chunk-R4UWBC35.js +200 -0
- package/dist/chunk-R6W23LKN.js +254 -0
- package/dist/chunk-RAK5HOIJ.js +459 -0
- package/dist/chunk-RDFRCT64.js +168 -0
- package/dist/chunk-RE5VZDFQ.js +58 -0
- package/dist/chunk-RGROYC2G.js +638 -0
- package/dist/chunk-RGWHUI5E.js +68 -0
- package/dist/chunk-RHVG6UNL.js +205 -0
- package/dist/chunk-RI4EX2QE.js +18 -0
- package/dist/chunk-RIKFAWZF.js +71 -0
- package/dist/chunk-RJAAKDRX.js +17 -0
- package/dist/chunk-RSXSD4MH.js +1016 -0
- package/dist/chunk-RW7RVSDV.js +79 -0
- package/dist/chunk-RWREC3KJ.js +517 -0
- package/dist/chunk-RZY7RKM5.js +500 -0
- package/dist/chunk-S2MNWFAG.js +52 -0
- package/dist/chunk-S2VQCZO4.js +22 -0
- package/dist/chunk-S54MKU6V.js +117 -0
- package/dist/chunk-S77XPALC.js +239 -0
- package/dist/chunk-SCW4ZF6R.js +150 -0
- package/dist/chunk-SKHGYQ7W.js +405 -0
- package/dist/chunk-SNLBO4BF.js +247 -0
- package/dist/chunk-SP3LFU6B.js +60 -0
- package/dist/chunk-SQZRPY23.js +5181 -0
- package/dist/chunk-SSB5UQHA.js +4128 -0
- package/dist/chunk-STJOHHSF.js +605 -0
- package/dist/chunk-STV6LYYE.js +45 -0
- package/dist/chunk-SUN4OFTW.js +740 -0
- package/dist/chunk-SURZ2WFE.js +326 -0
- package/dist/chunk-SUZ7UYZH.js +1972 -0
- package/dist/chunk-SZ7DL357.js +1047 -0
- package/dist/chunk-T2SZELJL.js +624 -0
- package/dist/chunk-T6VPI2XP.js +10338 -0
- package/dist/chunk-TAJPCXSB.js +13 -0
- package/dist/chunk-TAOOYK3P.js +358 -0
- package/dist/chunk-TDSEX5CF.js +24 -0
- package/dist/chunk-TFTGP7Z7.js +174 -0
- package/dist/chunk-TGON4N4O.js +197 -0
- package/dist/chunk-TIBZPAD2.js +9924 -0
- package/dist/chunk-TK644IOG.js +1813 -0
- package/dist/chunk-TRSVUPCX.js +25 -0
- package/dist/chunk-TSPQONRT.js +4533 -0
- package/dist/chunk-TSXGRLPR.js +107 -0
- package/dist/chunk-TWYLVAS2.js +152 -0
- package/dist/chunk-U2S3YEX3.js +224 -0
- package/dist/chunk-U32HWI76.js +437 -0
- package/dist/chunk-U45RTRGY.js +818 -0
- package/dist/chunk-U62VCQGR.js +49 -0
- package/dist/chunk-UDMS5AD3.js +307 -0
- package/dist/chunk-UGX7GI37.js +94 -0
- package/dist/chunk-UH3JJSHK.js +1 -0
- package/dist/chunk-UHRBTZYY.js +208 -0
- package/dist/chunk-UHT2KRG5.js +193 -0
- package/dist/chunk-UI2F6DJ5.js +118 -0
- package/dist/chunk-UKMAFG7A.js +189 -0
- package/dist/chunk-UUH2RVKS.js +585 -0
- package/dist/chunk-UVBBFDFQ.js +270 -0
- package/dist/chunk-UXZF3B7T.js +75 -0
- package/dist/chunk-V6YLL5WA.js +3746 -0
- package/dist/chunk-V72RMBM4.js +48 -0
- package/dist/chunk-VBNCKCCI.js +16 -0
- package/dist/chunk-VCAUV5XQ.js +3679 -0
- package/dist/chunk-VDSH7O4Y.js +152 -0
- package/dist/chunk-VNSYKC25.js +24 -0
- package/dist/chunk-VTLNJQ44.js +133 -0
- package/dist/chunk-VV6HOXRC.js +302 -0
- package/dist/chunk-VVGEPBPS.js +218 -0
- package/dist/chunk-VXF4Q3FW.js +20 -0
- package/dist/chunk-W3ZGW3C5.js +331 -0
- package/dist/chunk-W4OMESPG.js +271 -0
- package/dist/chunk-WDUORIHF.js +877 -0
- package/dist/chunk-WDVTY4U6.js +148 -0
- package/dist/chunk-WGXYDLMX.js +4386 -0
- package/dist/chunk-WMPU2UOV.js +82 -0
- package/dist/chunk-WOEVVPDP.js +1125 -0
- package/dist/chunk-WTLMSPBQ.js +256 -0
- package/dist/chunk-X43NMU4Z.js +1124 -0
- package/dist/chunk-X5NTI5U5.js +368 -0
- package/dist/chunk-X6UIEADS.js +949 -0
- package/dist/chunk-XAONGNST.js +57 -0
- package/dist/chunk-XB5YLCLB.js +13 -0
- package/dist/chunk-XGYF2QMZ.js +351 -0
- package/dist/chunk-XHEKQENB.js +255 -0
- package/dist/chunk-XPHKYHO4.js +878 -0
- package/dist/chunk-XQD2B6BB.js +507 -0
- package/dist/chunk-XT2ZQVTC.js +558 -0
- package/dist/chunk-XUSRFFA7.js +1859 -0
- package/dist/chunk-XUXVDPSZ.js +50 -0
- package/dist/chunk-XV7NZL4B.js +57 -0
- package/dist/chunk-XXCGJGSQ.js +121 -0
- package/dist/chunk-XYGDZCB5.js +659 -0
- package/dist/chunk-XZAMZUX5.js +992 -0
- package/dist/chunk-XZE4XARH.js +563 -0
- package/dist/chunk-Y34DUC4A.js +1356 -0
- package/dist/chunk-Y3J4JMTU.js +61 -0
- package/dist/chunk-Y4RC2YFX.js +1270 -0
- package/dist/chunk-Y55LZN5U.js +52 -0
- package/dist/chunk-Y5677NDO.js +183 -0
- package/dist/chunk-YESZJFK6.js +220 -0
- package/dist/chunk-YGXKOBQQ.js +292 -0
- package/dist/chunk-YMX76IOS.js +31 -0
- package/dist/chunk-YOTXESEH.js +257 -0
- package/dist/chunk-YQAAKTVH.js +712 -0
- package/dist/chunk-YVORHQ2S.js +579 -0
- package/dist/chunk-YWNAHW24.js +1552 -0
- package/dist/chunk-Z476RATD.js +1689 -0
- package/dist/chunk-ZDTGECTN.js +235 -0
- package/dist/chunk-ZDTMICNY.js +316 -0
- package/dist/chunk-ZIPWI2LY.js +151 -0
- package/dist/chunk-ZIVYXF7A.js +2344 -0
- package/dist/chunk-ZX65UI5W.js +78 -0
- package/dist/chunk-ZYG7GGKO.js +218 -0
- package/dist/chunk-ZZZUL4OF.js +38 -0
- package/dist/cli-utils-HUOTBRM3.js +40 -0
- package/dist/cli-validation-WLXWLUJL.js +111 -0
- package/dist/commit-utils-3QV2P6RA.js +202 -0
- package/dist/compete-config-resolver-SNH3TT2Q.js +76 -0
- package/dist/compete-request-XWNPOARL.js +364 -0
- package/dist/compete-results-PAP67LDR.js +208 -0
- package/dist/competition-outcome-tracker-GOZXNAQU.js +38 -0
- package/dist/config-HWHUNA72.js +166 -0
- package/dist/config-audit-EL3GHXS7.js +360 -0
- package/dist/config-explain-TNILDEVJ.js +12 -0
- package/dist/config-loader-VPFPFO4B.js +38 -0
- package/dist/config-manager-L3VRNTSG.js +67 -0
- package/dist/config-validator-SIIQEOFQ.js +72 -0
- package/dist/cost-7CX37DU3.js +53 -0
- package/dist/coverage-orchestrator-MNKHD6YC.js +772 -0
- package/dist/coverage-scanner-YVD62MLH.js +12 -0
- package/dist/createCompetitionRunner-LJG6ZQSO.js +46 -0
- package/dist/data-migration-state-VJK64DLT.js +31 -0
- package/dist/db-migrator-XSEIOGWY.js +34 -0
- package/dist/discovery-JGYGDD25.js +25 -0
- package/dist/doctor-CGBCD7IQ.js +186 -0
- package/dist/domain-analyzer-DL6BDR2N.js +29 -0
- package/dist/ensure-initialized-4BY73HLE.js +52 -0
- package/dist/entry.js +940 -0
- package/dist/environment-IGVH5X53.js +81 -0
- package/dist/error-parser-core-PPZ36A4S.js +50 -0
- package/dist/escalation-ladder-NWY5BJ57.js +130 -0
- package/dist/evaluation-context-KMWDSDYD.js +35 -0
- package/dist/evidence-plan-resolver-YIVH76PU.js +85 -0
- package/dist/execution-data-writer-OHAQ7G5H.js +53 -0
- package/dist/execution-policy-applier-ZHD3DCF3.js +31 -0
- package/dist/execution-preflight-Z4Y64V3I.js +209 -0
- package/dist/execution-roles-KBH2MC3M.js +66 -0
- package/dist/exit-code-error-Z4SW2DVK.js +14 -0
- package/dist/external-temp-cleanup-N2RZH4QP.js +311 -0
- package/dist/factory-IYBMCZG7.js +149 -0
- package/dist/file-utils-XIVR2ZMA.js +47 -0
- package/dist/flywheel-autoloop-executor-R3KD2G4D.js +249 -0
- package/dist/flywheel-competition-executor-XCFGVJSK.js +259 -0
- package/dist/flywheel-executor-LTSIZN2L.js +369 -0
- package/dist/flywheel-git-isolation-CZ5YVVZY.js +48 -0
- package/dist/flywheel-manifest-5UC4VBMA.js +108 -0
- package/dist/flywheel-model-defaults-7CVKP53V.js +27 -0
- package/dist/flywheel-preflight-6E33DJDK.js +104 -0
- package/dist/flywheel-resume-7HELH7QS.js +23 -0
- package/dist/flywheel-safety-7U7DPOAP.js +28 -0
- package/dist/flywheel-scope-decision-ZHKF6ZIN.js +11 -0
- package/dist/get-telemetry-logger-AZ46EPRP.js +73 -0
- package/dist/git-worktree-IY6V6DUO.js +18 -0
- package/dist/github-pr-manager-4XUO2YTM.js +211 -0
- package/dist/guidance-profile-config-E7XJL2VP.js +117 -0
- package/dist/guidance-profile-renderer-QK2YMAAP.js +21 -0
- package/dist/index-query-4JEUKA44.js +31 -0
- package/dist/index.js +108644 -0
- package/dist/instructions-CM6THDV7.js +64 -0
- package/dist/invoke-provider-auth-V5TC5UDS.js +223 -0
- package/dist/judge-scoring.json +80 -0
- package/dist/judge-test.js +555 -0
- package/dist/lab-mode-OTWVHN63.js +39 -0
- package/dist/lab-utils-VA6VXNOV.js +23 -0
- package/dist/linux-host-class-UY4KVLFD.js +29 -0
- package/dist/llm-judge-executor-34YPIRJ7.js +133 -0
- package/dist/local-executor-LJJSCWZX.js +223 -0
- package/dist/logger-6PRYM56M.js +72 -0
- package/dist/loop-spec-XTLQAVZ7.js +191 -0
- package/dist/manager-2HJG5CLT.js +25 -0
- package/dist/matrix-prompt-builder-LHDTJ4TY.js +497 -0
- package/dist/matrix-result-parser-7W35EUOV.js +15 -0
- package/dist/metadata-HQ43NDEK.js +22 -0
- package/dist/model-invocations-db-QBO4Y7WZ.js +38 -0
- package/dist/monitor-MCI3F4PX.js +72 -0
- package/dist/next-HQKDXIIR.js +90 -0
- package/dist/next-JC3V2UCS.js +69 -0
- package/dist/optimization-config-resolver-QPSFLCJU.js +75 -0
- package/dist/package-3YCWIQQ5.js +10 -0
- package/dist/parallel-orchestrator-NL7YJJU2.js +297 -0
- package/dist/parallel-worker.js +279 -0
- package/dist/parse-git-status-PIAF3OTP.js +10 -0
- package/dist/path-security-ETTH6TZL.js +39 -0
- package/dist/pattern-bank-UNJR5ADV.js +15 -0
- package/dist/plan-backlog-list-fast-DZK24LSZ.js +270 -0
- package/dist/plan-backlog-show-fast-63GBBH6Q.js +406 -0
- package/dist/plan-sprint-list-fast-TJ6ZUF5E.js +162 -0
- package/dist/plan-sprint-status-fast-YWE5X3NG.js +208 -0
- package/dist/plan-task-status-fast-K22BTTBW.js +349 -0
- package/dist/plan-to-flywheel-VFGCPINA.js +456 -0
- package/dist/planning-backlog-Y6TIJYWU.js +319 -0
- package/dist/poetic-root-V5DNXEAE.js +17 -0
- package/dist/preflight-OPANWMME.js +12 -0
- package/dist/process-registry-PDZJPTAS.js +27 -0
- package/dist/process-scanner-C22SYXYD.js +22 -0
- package/dist/processes-KMQUD7HY.js +29 -0
- package/dist/processes-json-fast-IBDXZ6WW.js +135 -0
- package/dist/profile-2XP3FWH2.js +149 -0
- package/dist/prompts-EPAAAMMF.js +168 -0
- package/dist/protected-branches-RNTEX7ET.js +57 -0
- package/dist/provider-aware-resource-manager-QBTZJMNP.js +18 -0
- package/dist/provider-matrix-KWTYA5QV.js +154 -0
- package/dist/provider-matrix-renderer-MZDWKKID.js +139 -0
- package/dist/provider-output-forwarding-DERID3TP.js +30 -0
- package/dist/provider-registry-J5NRBJKI.js +128 -0
- package/dist/provider-temp-cleanup-KYGLF6YU.js +14 -0
- package/dist/prune-engine-Q6E2DXBN.js +451 -0
- package/dist/quality-gate-33SHK43X.js +70 -0
- package/dist/quickstart-E32SODTA.js +62 -0
- package/dist/readiness-REWW7OSV.js +96 -0
- package/dist/registry-N6L36RE2.js +153 -0
- package/dist/repair-helpers-TTV5FEF4.js +117 -0
- package/dist/repo-root-2IGD4H7X.js +22 -0
- package/dist/resolution-engine-LXQYDSKS.js +127 -0
- package/dist/resolve-provider-cli-O2P6XZBZ.js +35 -0
- package/dist/restore-telemetry-db-BQL6DTQF.js +360 -0
- package/dist/result-streamer-WEHW476R.js +19 -0
- package/dist/routing-events-DHPUZDDF.js +27 -0
- package/dist/routing-history-store-UWCEXAP3.js +46 -0
- package/dist/routing-planner-VGQU35QM.js +160 -0
- package/dist/run-poetic-tui-ZZXFDOQ4.js +13517 -0
- package/dist/run-simulation.js +376 -0
- package/dist/runner-2YPKOJ6A.js +1092 -0
- package/dist/runner-HEJJ65AJ.js +1571 -0
- package/dist/safety-OCC4KJJR.js +64 -0
- package/dist/safety-stanza-BSVE6OJU.js +60 -0
- package/dist/sandbox-3AG2WJLM.js +94 -0
- package/dist/sandbox-5JOFBS75.js +81 -0
- package/dist/schema-extensions.sql +234 -0
- package/dist/security-2KI5XNDG.js +17 -0
- package/dist/setup-bb-plugin-QAQWZCV5.js +738 -0
- package/dist/setup-claude-code-P7BTTSPO.js +317 -0
- package/dist/setup-temp-cleanup-GASCWZ2I.js +20 -0
- package/dist/shared-utils-D6U6UREP.js +37 -0
- package/dist/simple-artifact-goal-OTXNGZY6.js +25 -0
- package/dist/sprint-QGXLN3SO.js +338 -0
- package/dist/sprint-bridge-FNKQCJGL.js +68 -0
- package/dist/sprint-execution-service-4ELKNJDS.js +311 -0
- package/dist/sprint-manager-7TNCXCIX.js +91 -0
- package/dist/sqlite-wrapper-UDWCTNLJ.js +17 -0
- package/dist/stale-cleanup-orchestrator-H5XKGZP2.js +62 -0
- package/dist/storage-access-F6ALDN4S.js +22 -0
- package/dist/synthesis-TGQUVAW3.js +185 -0
- package/dist/task-classifier-DZIRBSS2.js +494 -0
- package/dist/task-list-fast-PJ2EXVZE.js +147 -0
- package/dist/task-manager-5742GMJK.js +93 -0
- package/dist/task-splitter-6MBXSWF6.js +79 -0
- package/dist/task-type-detector-OLVOMXZO.js +21 -0
- package/dist/task-type-resolver-4HG6TXFV.js +42 -0
- package/dist/telemetry-JX7EMNOZ.js +131 -0
- package/dist/telemetry-health-tracker-DMSSGBZ2.js +11 -0
- package/dist/telemetry-path-resolver-7ICSG276.js +28 -0
- package/dist/telemetry-reliability-2MLJ4J6B.js +125 -0
- package/dist/token-estimator-PDU345YE.js +37 -0
- package/dist/transports-3UEO6C5Y.js +255 -0
- package/dist/unified-cost-tracker-F3W5NMPG.js +35 -0
- package/dist/unified-state-cleanup-LJPF2BOW.js +276 -0
- package/dist/universal-cost-calculator-BAJV35GA.js +42 -0
- package/dist/user-agents-GHUGCSKN.js +22 -0
- package/dist/variant-delivery-state-XCOYVCM4.js +33 -0
- package/dist/variant-state-cleanup-QIEKRULB.js +280 -0
- package/dist/variant-success-PJXWE3ZV.js +12 -0
- package/dist/variant-worker-HPTUA37Z.js +4196 -0
- package/dist/variant-worker.js +24 -0
- package/dist/variants-2HDYZOCG.js +233 -0
- package/dist/verification-toolchains-VVKT3GBF.js +28 -0
- package/dist/verify-commands-5F6CJV4B.js +37 -0
- package/dist/warning-EGQGUUIO.js +12 -0
- package/dist/work-edge-RWEDWJZF.js +49 -0
- package/dist/work-item-writer-WHAVZXAY.js +19 -0
- package/dist/worker-AGWHD4D2.js +1435 -0
- package/dist/worker.js +19 -0
- package/dist/worktree-metrics-3DFUYVYE.js +26 -0
- package/dist/writeback-VZKHYHSN.js +44 -0
- package/docs/CLI_REFERENCE.md +7040 -0
- package/docs/PROVIDER_SETUP.md +1570 -0
- package/docs/README.md +96 -0
- package/docs/TROUBLESHOOTING.md +2232 -0
- package/docs/getting-started/QUICK_START.md +294 -0
- package/docs/getting-started/README.md +188 -0
- package/docs/getting-started/SETUP.md +52 -0
- package/docs/reference/AGENT_CONTEXT.md +38 -0
- package/docs/reference/PROVIDER_RELEASE_TIERS.md +34 -0
- package/docs/reference/README.md +381 -0
- package/docs/reference/SECURITY.md +262 -0
- package/integrations/bb-plugin-poetic/README.md +362 -0
- package/integrations/bb-plugin-poetic/app.css +341 -0
- package/integrations/bb-plugin-poetic/app.tsx +5059 -0
- package/integrations/bb-plugin-poetic/package.json +45 -0
- package/integrations/bb-plugin-poetic/server.ts +444 -0
- package/integrations/bb-plugin-poetic/src/adapter.ts +3312 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-model.ts +150 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-schema.ts +355 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-service.ts +419 -0
- package/integrations/bb-plugin-poetic/src/backlog-authoring-view.ts +1026 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-schema.ts +122 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-service.ts +105 -0
- package/integrations/bb-plugin-poetic/src/competition-defaults-view.ts +138 -0
- package/integrations/bb-plugin-poetic/src/contract.ts +615 -0
- package/integrations/bb-plugin-poetic/src/finalize-model.ts +205 -0
- package/integrations/bb-plugin-poetic/src/finalize-schema.ts +158 -0
- package/integrations/bb-plugin-poetic/src/finalize-service.ts +382 -0
- package/integrations/bb-plugin-poetic/src/finalize-view.ts +159 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-readback.ts +303 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-schema.ts +33 -0
- package/integrations/bb-plugin-poetic/src/judge-operations-view.ts +459 -0
- package/integrations/bb-plugin-poetic/src/model.ts +1302 -0
- package/integrations/bb-plugin-poetic/src/monitor-service.ts +233 -0
- package/integrations/bb-plugin-poetic/src/panel-read-ux.ts +97 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-schema.ts +166 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-service.ts +480 -0
- package/integrations/bb-plugin-poetic/src/patch-preview-view.ts +109 -0
- package/integrations/bb-plugin-poetic/src/planning-readback.ts +204 -0
- package/integrations/bb-plugin-poetic/src/planning-service.ts +594 -0
- package/integrations/bb-plugin-poetic/src/planning-workspace-schema.ts +95 -0
- package/integrations/bb-plugin-poetic/src/planning-workspace.ts +323 -0
- package/integrations/bb-plugin-poetic/src/poll-handoff.ts +136 -0
- package/integrations/bb-plugin-poetic/src/project-target-server.ts +21 -0
- package/integrations/bb-plugin-poetic/src/project-target.ts +45 -0
- package/integrations/bb-plugin-poetic/src/reference-index.ts +192 -0
- package/integrations/bb-plugin-poetic/src/repository-schema.ts +30 -0
- package/integrations/bb-plugin-poetic/src/result-explorer-view.ts +597 -0
- package/integrations/bb-plugin-poetic/src/result-readback-model.ts +163 -0
- package/integrations/bb-plugin-poetic/src/result-readback-schema.ts +319 -0
- package/integrations/bb-plugin-poetic/src/result-readback-service.ts +333 -0
- package/integrations/bb-plugin-poetic/src/run-page-view.ts +177 -0
- package/integrations/bb-plugin-poetic/src/setup-config-schema.ts +291 -0
- package/integrations/bb-plugin-poetic/src/setup-config-service.ts +590 -0
- package/integrations/bb-plugin-poetic/src/setup-config-view.ts +413 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-model.ts +150 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-schema.ts +136 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-service.ts +341 -0
- package/integrations/bb-plugin-poetic/src/sprint-close-view.ts +116 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-model.ts +714 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-readback.ts +231 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-schema.ts +601 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-service.ts +775 -0
- package/integrations/bb-plugin-poetic/src/sprint-composition-view.ts +1221 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-model.ts +567 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-schema.ts +635 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-service.ts +842 -0
- package/integrations/bb-plugin-poetic/src/task-authoring-view.ts +1227 -0
- package/integrations/bb-plugin-poetic/src/task-detail-readback.ts +56 -0
- package/integrations/bb-plugin-poetic/src/task-detail-schema.ts +144 -0
- package/integrations/bb-plugin-poetic/src/task-navigation.ts +446 -0
- package/integrations/bb-plugin-poetic/src/workbench-route.ts +74 -0
- package/integrations/bb-plugin-poetic/tests/host-contract.test.ts +773 -0
- package/integrations/bb-plugin-poetic/tsconfig.json +18 -0
- package/integrations/bb-plugin-poetic/types/PROVENANCE.json +52 -0
- package/integrations/bb-plugin-poetic/types/bb-plugin-sdk-app.d.ts +1444 -0
- package/integrations/bb-plugin-poetic/types/bb-plugin-sdk.d.ts +13030 -0
- package/integrations/bb-plugin-poetic/vitest.host.config.ts +71 -0
- package/package.json +331 -0
- package/schemas/README.md +80 -0
- package/schemas/config-v1.schema.json +136 -0
- package/schemas/execution-config.schema.json +47 -0
- package/schemas/judge-scoring.schema.json +370 -0
- package/schemas/poetic.config.schema.json +2214 -0
- package/schemas/provider-config.schema.json +203 -0
- package/schemas/telemetry-config.schema.json +79 -0
- package/schemas/user-preferences.schema.json +141 -0
- package/scripts/assert-node-runtime.mjs +140 -0
- package/scripts/preflight-native.mjs +49 -0
- package/scripts/preinstall-node-check.mjs +78 -0
- package/scripts/setup-git-hooks.mjs +24 -0
- package/scripts/sync.sh +2722 -0
- package/scripts/write-node-launcher.sh +108 -0
- package/src/resources/gemini/slash-packs/default/plan.toml +15 -0
- package/src/resources/gemini/slash-packs/default/summary.toml +16 -0
- package/src/resources/gemini/slash-packs/default/tests.toml +16 -0
- package/templates/.poetic/README.md +37 -0
- package/templates/.poetic/agents/README.md +296 -0
- package/templates/.poetic/agents/api-documenter.md +147 -0
- package/templates/.poetic/agents/backend-architect.md +31 -0
- package/templates/.poetic/agents/code-reviewer.md +157 -0
- package/templates/.poetic/agents/data-scientist.md +179 -0
- package/templates/.poetic/agents/database-optimizer.md +145 -0
- package/templates/.poetic/agents/debugger.md +31 -0
- package/templates/.poetic/agents/deployment-engineer.md +164 -0
- package/templates/.poetic/agents/devops-troubleshooter.md +139 -0
- package/templates/.poetic/agents/frontend-developer.md +150 -0
- package/templates/.poetic/agents/javascript-pro.md +36 -0
- package/templates/.poetic/agents/performance-engineer.md +151 -0
- package/templates/.poetic/agents/python-pro.md +137 -0
- package/templates/.poetic/agents/test-automator.md +147 -0
- package/templates/.poetic/agents/typescript-pro.md +34 -0
- package/templates/.poetic/config/poetic.config.jsonc +69 -0
- package/templates/.poetic/config/project-context.template.json +6 -0
- package/templates/.poetic/config/task-type-aliases.presets/kanban.yaml +14 -0
- package/templates/.poetic/config/task-type-aliases.presets/scrum.yaml +17 -0
- package/templates/.poetic/config/task-type-aliases.presets/xp.yaml +12 -0
- package/templates/.poetic/config/task-type-aliases.yaml +28 -0
- package/templates/.poetic/gitignore.template +55 -0
- package/templates/.poetic/task-types/analysis.yaml +40 -0
- package/templates/.poetic/task-types/architecture.yaml +38 -0
- package/templates/.poetic/task-types/doc.yaml +35 -0
- package/templates/.poetic/task-types/feature.yaml +23 -0
- package/templates/.poetic/task-types/general.yaml +6 -0
- package/templates/.poetic/task-types/security.yaml +39 -0
- package/templates/AGENTS.template.md +99 -0
- package/templates/CLAUDE.template.md +1 -0
- package/templates/GEMINI.template.md +1 -0
- package/templates/builtin-workflows/code-review.yaml +73 -0
- package/templates/builtin-workflows/compete-streak.yaml +78 -0
- package/templates/builtin-workflows/hello-verify.yaml +10 -0
- package/templates/builtin-workflows/judge-regression.yaml +114 -0
- package/templates/builtin-workflows/skills/code-review/SKILL.md +60 -0
- package/templates/builtin-workflows/skills/hello-verify/SKILL.md +6 -0
- package/templates/guard-kit/GUARD_SETUP.md.template +255 -0
- package/templates/guard-kit/check.mjs.template +777 -0
- package/templates/guard-kit/config.json.template +6 -0
- package/templates/guard-kit/poetic-guard.yml.template +189 -0
- package/templates/profiles/README.md +56 -0
- package/templates/profiles/frontier-claude.json +27 -0
- package/templates/profiles/frontier-codex.json +26 -0
|
@@ -0,0 +1,4533 @@
|
|
|
1
|
+
import {
|
|
2
|
+
charsToTokens
|
|
3
|
+
} from "./chunk-LFTQVYBB.js";
|
|
4
|
+
import {
|
|
5
|
+
runLintOnDiffWithArtifacts,
|
|
6
|
+
runTestGate
|
|
7
|
+
} from "./chunk-EJ42BNGY.js";
|
|
8
|
+
import {
|
|
9
|
+
DETERMINISTIC_WINNER_WEIGHTS,
|
|
10
|
+
THREE_BUCKET_RUBRIC_FORMULA_TEXT,
|
|
11
|
+
THREE_BUCKET_RUBRIC_PERCENT_TEXT
|
|
12
|
+
} from "./chunk-73NCTWKL.js";
|
|
13
|
+
import {
|
|
14
|
+
formatVerificationEvidenceQualitySummary
|
|
15
|
+
} from "./chunk-E74LUIID.js";
|
|
16
|
+
import {
|
|
17
|
+
formatVerificationProvenanceForPrompt,
|
|
18
|
+
resolveJudgeEvaluationContext
|
|
19
|
+
} from "./chunk-66AGFTBQ.js";
|
|
20
|
+
import {
|
|
21
|
+
extractTaskContract
|
|
22
|
+
} from "./chunk-AZQILSNQ.js";
|
|
23
|
+
import {
|
|
24
|
+
categorizeFile,
|
|
25
|
+
extractChangedFilesMeta
|
|
26
|
+
} from "./chunk-CFBIG37O.js";
|
|
27
|
+
import {
|
|
28
|
+
hasRecordedTestExecutionEvidence,
|
|
29
|
+
hasTestExecutionProvenance,
|
|
30
|
+
hasUnrecognizedRecordedTestStatus,
|
|
31
|
+
normalizeRecordedTestStatus
|
|
32
|
+
} from "./chunk-KT6VHSPF.js";
|
|
33
|
+
import {
|
|
34
|
+
taskTypeSignalsEnabled
|
|
35
|
+
} from "./chunk-W4OMESPG.js";
|
|
36
|
+
import {
|
|
37
|
+
parsePromptSpec
|
|
38
|
+
} from "./chunk-X6UIEADS.js";
|
|
39
|
+
import {
|
|
40
|
+
parseDiffGitHeaderPaths
|
|
41
|
+
} from "./chunk-C3L6YQ7P.js";
|
|
42
|
+
import {
|
|
43
|
+
runBuildGate
|
|
44
|
+
} from "./chunk-STJOHHSF.js";
|
|
45
|
+
import {
|
|
46
|
+
resolveVerificationChannelScope
|
|
47
|
+
} from "./chunk-MCPQNJMK.js";
|
|
48
|
+
import {
|
|
49
|
+
resolveJudgeConfig
|
|
50
|
+
} from "./chunk-IZZK3H6I.js";
|
|
51
|
+
import {
|
|
52
|
+
parseVariantIdentifier
|
|
53
|
+
} from "./chunk-YWNAHW24.js";
|
|
54
|
+
import {
|
|
55
|
+
escapeEvidenceForJudgePromptDisplay,
|
|
56
|
+
escapePathForJudgePromptDisplay,
|
|
57
|
+
extractCleanStdout,
|
|
58
|
+
extractDiffText,
|
|
59
|
+
frameUntrustedJudgeEvidence,
|
|
60
|
+
frameUntrustedJudgeTaskText,
|
|
61
|
+
presentUntrustedJudgeEvidence,
|
|
62
|
+
redactSuspiciousJudgePromptPatterns
|
|
63
|
+
} from "./chunk-IATQ7NBI.js";
|
|
64
|
+
import {
|
|
65
|
+
JUDGE_FEATURE_FLAGS
|
|
66
|
+
} from "./chunk-BSE4R6XD.js";
|
|
67
|
+
import {
|
|
68
|
+
isProjectConfigPath,
|
|
69
|
+
isProjectTestPath,
|
|
70
|
+
languageForProjectPath
|
|
71
|
+
} from "./chunk-UDMS5AD3.js";
|
|
72
|
+
import {
|
|
73
|
+
describeVerificationCommand
|
|
74
|
+
} from "./chunk-4BIPZBT7.js";
|
|
75
|
+
import {
|
|
76
|
+
ProviderRegistry
|
|
77
|
+
} from "./chunk-AT4TNPWW.js";
|
|
78
|
+
import {
|
|
79
|
+
sanitizeString
|
|
80
|
+
} from "./chunk-LPITVM7M.js";
|
|
81
|
+
import {
|
|
82
|
+
POETIC_LAB_ID,
|
|
83
|
+
POETIC_LAB_MODE,
|
|
84
|
+
POETIC_LAB_RUN_ID,
|
|
85
|
+
readEnv
|
|
86
|
+
} from "./chunk-5DLKYTQX.js";
|
|
87
|
+
|
|
88
|
+
// src/core/judge/pre-score-gates.ts
|
|
89
|
+
import path from "path";
|
|
90
|
+
|
|
91
|
+
// src/core/judge/requirement-validator.ts
|
|
92
|
+
import { readFile } from "fs/promises";
|
|
93
|
+
import { join } from "path";
|
|
94
|
+
async function validateRequirements(requirements, variantPath, changedFiles) {
|
|
95
|
+
if (requirements.length === 0) {
|
|
96
|
+
return {
|
|
97
|
+
satisfied: true,
|
|
98
|
+
satisfiedCount: 0,
|
|
99
|
+
totalCount: 0,
|
|
100
|
+
missingRequirements: [],
|
|
101
|
+
details: []
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
const details = [];
|
|
105
|
+
let satisfiedCount = 0;
|
|
106
|
+
for (const requirement of requirements) {
|
|
107
|
+
let result;
|
|
108
|
+
switch (requirement.type) {
|
|
109
|
+
case "cli-flag":
|
|
110
|
+
result = {
|
|
111
|
+
...await validateCliFlag(
|
|
112
|
+
requirement.metadata?.flagName ?? "",
|
|
113
|
+
variantPath,
|
|
114
|
+
changedFiles
|
|
115
|
+
),
|
|
116
|
+
verificationMethod: "cli_flag_scan",
|
|
117
|
+
verificationDepth: "structural"
|
|
118
|
+
};
|
|
119
|
+
break;
|
|
120
|
+
case "file-constraint":
|
|
121
|
+
result = {
|
|
122
|
+
...validateFileConstraint(
|
|
123
|
+
requirement.metadata?.files ?? [],
|
|
124
|
+
requirement.metadata?.mode ?? "soft",
|
|
125
|
+
changedFiles
|
|
126
|
+
),
|
|
127
|
+
verificationMethod: "file_target_match",
|
|
128
|
+
verificationDepth: "structural"
|
|
129
|
+
};
|
|
130
|
+
break;
|
|
131
|
+
case "tests":
|
|
132
|
+
result = {
|
|
133
|
+
...validateTestsRequirement(changedFiles),
|
|
134
|
+
verificationMethod: "test_file_presence",
|
|
135
|
+
verificationDepth: "lexical"
|
|
136
|
+
};
|
|
137
|
+
break;
|
|
138
|
+
case "feature":
|
|
139
|
+
case "explicit":
|
|
140
|
+
result = {
|
|
141
|
+
...await validateFeatureRequirement(requirement.description, variantPath, changedFiles),
|
|
142
|
+
verificationMethod: "keyword_match",
|
|
143
|
+
verificationDepth: "lexical"
|
|
144
|
+
};
|
|
145
|
+
break;
|
|
146
|
+
default:
|
|
147
|
+
result = {
|
|
148
|
+
satisfied: false,
|
|
149
|
+
reason: "Unknown requirement type",
|
|
150
|
+
verificationMethod: "unknown",
|
|
151
|
+
verificationDepth: "lexical"
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
if (result.satisfied) {
|
|
155
|
+
satisfiedCount++;
|
|
156
|
+
}
|
|
157
|
+
details.push({
|
|
158
|
+
requirement,
|
|
159
|
+
satisfied: result.satisfied,
|
|
160
|
+
reason: result.reason,
|
|
161
|
+
verificationMethod: result.verificationMethod,
|
|
162
|
+
verificationDepth: result.verificationDepth
|
|
163
|
+
});
|
|
164
|
+
}
|
|
165
|
+
const missingRequirements = details.filter((d) => !d.satisfied).map((d) => d.requirement.description);
|
|
166
|
+
return {
|
|
167
|
+
satisfied: satisfiedCount === requirements.length,
|
|
168
|
+
satisfiedCount,
|
|
169
|
+
totalCount: requirements.length,
|
|
170
|
+
missingRequirements,
|
|
171
|
+
details
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
function validateTestsRequirement(changedFiles) {
|
|
175
|
+
const testFiles = changedFiles.filter((f) => isProjectTestPath(f));
|
|
176
|
+
if (testFiles.length === 0) {
|
|
177
|
+
return { satisfied: false, reason: "No test files were added or modified" };
|
|
178
|
+
}
|
|
179
|
+
return {
|
|
180
|
+
satisfied: true,
|
|
181
|
+
reason: `Modified ${testFiles.length} test file(s): ${testFiles.slice(0, 3).join(", ")}`
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
async function validateCliFlag(flagName, variantPath, changedFiles) {
|
|
185
|
+
if (!flagName) {
|
|
186
|
+
return { satisfied: false, reason: "No flag name provided" };
|
|
187
|
+
}
|
|
188
|
+
const cliFiles = changedFiles.filter(
|
|
189
|
+
(f) => f.includes("src/cli/") || f.toLowerCase().includes("option") || f.toLowerCase().includes("handler") || f.toLowerCase().includes("config")
|
|
190
|
+
);
|
|
191
|
+
if (cliFiles.length === 0) {
|
|
192
|
+
return {
|
|
193
|
+
satisfied: false,
|
|
194
|
+
reason: `No CLI-related files modified for flag --${flagName}`
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
let definitionFound = false;
|
|
198
|
+
let definitionReason = "";
|
|
199
|
+
const definitionFailedReads = [];
|
|
200
|
+
for (const file of cliFiles) {
|
|
201
|
+
try {
|
|
202
|
+
const filePath = join(variantPath, file);
|
|
203
|
+
const content = await readFile(filePath, "utf-8");
|
|
204
|
+
const camelFlagName = flagName.replace(/-([a-z])/g, (_, letter) => letter.toUpperCase());
|
|
205
|
+
const flagPatterns = [
|
|
206
|
+
new RegExp(`['"\`]--${flagName}['"\`]`, "i"),
|
|
207
|
+
// '--flag-name'
|
|
208
|
+
new RegExp(`['"\`]${flagName}['"\`].*:`, "i"),
|
|
209
|
+
// 'flag-name':
|
|
210
|
+
new RegExp(`['"]${flagName}['"]\\s*:`, "i"),
|
|
211
|
+
// 'flag-name':
|
|
212
|
+
new RegExp(`${camelFlagName}\\s*:`, "i"),
|
|
213
|
+
// camelCase: flagName:
|
|
214
|
+
new RegExp(`option\\(['"\`]${flagName}['"\`]`, "i")
|
|
215
|
+
// option('flag-name')
|
|
216
|
+
];
|
|
217
|
+
if (flagPatterns.some((pattern) => pattern.test(content))) {
|
|
218
|
+
definitionFound = true;
|
|
219
|
+
definitionReason = `Flag defined in ${file}`;
|
|
220
|
+
break;
|
|
221
|
+
}
|
|
222
|
+
} catch (_error) {
|
|
223
|
+
definitionFailedReads.push(file);
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
if (definitionFailedReads.length > 0) {
|
|
227
|
+
console.warn(
|
|
228
|
+
`[requirement-validator] ${definitionFailedReads.length}/${cliFiles.length} files unreadable during CLI flag definition check (${definitionFailedReads.slice(0, 3).join(", ")})`
|
|
229
|
+
);
|
|
230
|
+
}
|
|
231
|
+
if (!definitionFound) {
|
|
232
|
+
return {
|
|
233
|
+
satisfied: false,
|
|
234
|
+
reason: `Flag --${flagName} definition not found in CLI files`
|
|
235
|
+
};
|
|
236
|
+
}
|
|
237
|
+
let propagationFound = false;
|
|
238
|
+
let propagationReason = "";
|
|
239
|
+
const propagationFailedReads = [];
|
|
240
|
+
for (const file of cliFiles) {
|
|
241
|
+
try {
|
|
242
|
+
const filePath = join(variantPath, file);
|
|
243
|
+
const content = await readFile(filePath, "utf-8");
|
|
244
|
+
const camelFlagName = flagName.replace(/-([a-z])/g, (_, letter) => letter.toUpperCase());
|
|
245
|
+
const usagePatterns = [
|
|
246
|
+
new RegExp(`options\\.${camelFlagName}`, "i"),
|
|
247
|
+
// options.maxRetries
|
|
248
|
+
new RegExp(`opts\\.${camelFlagName}`, "i"),
|
|
249
|
+
// opts.maxRetries
|
|
250
|
+
new RegExp(`\\[['"\`]${flagName}['"\`]\\]`, "i"),
|
|
251
|
+
// ['max-retries']
|
|
252
|
+
new RegExp(`get\\(['"\`]${flagName}['"\`]\\)`, "i"),
|
|
253
|
+
// get('max-retries')
|
|
254
|
+
new RegExp(`const\\s+${camelFlagName}\\s*=`, "i"),
|
|
255
|
+
// const maxRetries =
|
|
256
|
+
new RegExp(`\\b${camelFlagName}\\s*:`, "i")
|
|
257
|
+
// maxRetries: in object
|
|
258
|
+
];
|
|
259
|
+
if (usagePatterns.some((pattern) => pattern.test(content))) {
|
|
260
|
+
propagationFound = true;
|
|
261
|
+
propagationReason = `Flag propagated in ${file}`;
|
|
262
|
+
break;
|
|
263
|
+
}
|
|
264
|
+
} catch (_error) {
|
|
265
|
+
propagationFailedReads.push(file);
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
if (propagationFailedReads.length > 0) {
|
|
269
|
+
console.warn(
|
|
270
|
+
`[requirement-validator] ${propagationFailedReads.length}/${cliFiles.length} files unreadable during CLI flag propagation check (${propagationFailedReads.slice(0, 3).join(", ")})`
|
|
271
|
+
);
|
|
272
|
+
}
|
|
273
|
+
if (!propagationFound) {
|
|
274
|
+
return {
|
|
275
|
+
satisfied: false,
|
|
276
|
+
reason: `Flag --${flagName} defined but not propagated to handler`
|
|
277
|
+
};
|
|
278
|
+
}
|
|
279
|
+
return {
|
|
280
|
+
satisfied: true,
|
|
281
|
+
reason: `Flag --${flagName} implemented: ${definitionReason}, ${propagationReason}`
|
|
282
|
+
};
|
|
283
|
+
}
|
|
284
|
+
function validateFileConstraint(requiredFiles, mode, changedFiles) {
|
|
285
|
+
if (requiredFiles.length === 0) {
|
|
286
|
+
return { satisfied: true, reason: "No file constraints specified" };
|
|
287
|
+
}
|
|
288
|
+
const normalizedChangedFiles = changedFiles.map(
|
|
289
|
+
(f) => f.toLowerCase().replace(/^\.\//, "").replace(/\\/g, "/")
|
|
290
|
+
);
|
|
291
|
+
const normalizedTargets = requiredFiles.map(
|
|
292
|
+
(f) => f.toLowerCase().replace(/^\.\//, "").replace(/\\/g, "/")
|
|
293
|
+
);
|
|
294
|
+
if (mode === "hard") {
|
|
295
|
+
if (normalizedChangedFiles.length === 0) {
|
|
296
|
+
return { satisfied: false, reason: "No files changed under FILES_ONLY constraint" };
|
|
297
|
+
}
|
|
298
|
+
const extraFiles = normalizedChangedFiles.filter((f) => {
|
|
299
|
+
return !normalizedTargets.some((target) => {
|
|
300
|
+
if (f === target) return true;
|
|
301
|
+
if (target.endsWith("/") && f.startsWith(target)) return true;
|
|
302
|
+
if (!target.endsWith("/") && f.startsWith(`${target}/`)) return true;
|
|
303
|
+
if (target.includes("*")) {
|
|
304
|
+
const escaped = target.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
305
|
+
const regexStr = escaped.replace(/\*\*/g, ".*").replace(/\*/g, "[^/]*");
|
|
306
|
+
return new RegExp(`^${regexStr}$`).test(f);
|
|
307
|
+
}
|
|
308
|
+
return false;
|
|
309
|
+
});
|
|
310
|
+
});
|
|
311
|
+
if (extraFiles.length > 0) {
|
|
312
|
+
return {
|
|
313
|
+
satisfied: false,
|
|
314
|
+
reason: `Modified files outside FILES_ONLY targets: ${extraFiles.slice(0, 5).join(", ")}${extraFiles.length > 5 ? "..." : ""}`
|
|
315
|
+
};
|
|
316
|
+
}
|
|
317
|
+
return {
|
|
318
|
+
satisfied: true,
|
|
319
|
+
reason: `All changes contained within FILES_ONLY targets (${requiredFiles.length} targets)`
|
|
320
|
+
};
|
|
321
|
+
}
|
|
322
|
+
const modifiedTargets = requiredFiles.filter((requiredFile) => {
|
|
323
|
+
const normalized = requiredFile.toLowerCase().replace(/^\.\//, "").replace(/\\/g, "/");
|
|
324
|
+
if (normalizedChangedFiles.includes(normalized)) return true;
|
|
325
|
+
if (normalized.endsWith("/")) {
|
|
326
|
+
return normalizedChangedFiles.some((f) => f.startsWith(normalized));
|
|
327
|
+
}
|
|
328
|
+
if (normalized.includes("*")) {
|
|
329
|
+
const escaped = normalized.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
330
|
+
const regexStr = escaped.replace(/\*\*/g, ".*").replace(/\*/g, "[^/]*");
|
|
331
|
+
const regex = new RegExp(`^${regexStr}$`);
|
|
332
|
+
return normalizedChangedFiles.some((f) => regex.test(f));
|
|
333
|
+
}
|
|
334
|
+
return normalizedChangedFiles.some((f) => f.startsWith(`${normalized}/`));
|
|
335
|
+
});
|
|
336
|
+
if (modifiedTargets.length > 0) {
|
|
337
|
+
return {
|
|
338
|
+
satisfied: true,
|
|
339
|
+
reason: `Modified FILE target(s): ${modifiedTargets.join(", ")}`
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
return {
|
|
343
|
+
satisfied: false,
|
|
344
|
+
reason: `No FILE targets modified: ${requiredFiles.join(", ")}`
|
|
345
|
+
};
|
|
346
|
+
}
|
|
347
|
+
async function validateFeatureRequirement(description, variantPath, changedFiles) {
|
|
348
|
+
if (changedFiles.length === 0) {
|
|
349
|
+
return {
|
|
350
|
+
satisfied: false,
|
|
351
|
+
reason: "No files changed to implement feature"
|
|
352
|
+
};
|
|
353
|
+
}
|
|
354
|
+
const descriptionLower = (description ?? "").toLowerCase();
|
|
355
|
+
if (/\btests?\b/.test(descriptionLower)) {
|
|
356
|
+
const hasTestLikeChange = changedFiles.some((file) => isProjectTestPath(file));
|
|
357
|
+
if (hasTestLikeChange) {
|
|
358
|
+
return {
|
|
359
|
+
satisfied: true,
|
|
360
|
+
reason: "Test requirement satisfied by test-like file changes."
|
|
361
|
+
};
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
const keywords = extractKeywords(description);
|
|
365
|
+
if (keywords.length === 0) {
|
|
366
|
+
return {
|
|
367
|
+
satisfied: true,
|
|
368
|
+
reason: `Files changed: ${changedFiles.slice(0, 3).join(", ")}${changedFiles.length > 3 ? "..." : ""}`
|
|
369
|
+
};
|
|
370
|
+
}
|
|
371
|
+
const matchedKeywords = /* @__PURE__ */ new Set();
|
|
372
|
+
let readableFiles = 0;
|
|
373
|
+
let filenameMatchCount = 0;
|
|
374
|
+
const matchedFiles = [];
|
|
375
|
+
const filenameMatchedFiles = [];
|
|
376
|
+
const featureFailedReads = [];
|
|
377
|
+
for (const file of changedFiles.slice(0, 50)) {
|
|
378
|
+
const fileLower = file.toLowerCase();
|
|
379
|
+
const filePathMatches = keywords.some((keyword) => fileLower.includes(keyword.toLowerCase()));
|
|
380
|
+
if (filePathMatches) {
|
|
381
|
+
filenameMatchCount++;
|
|
382
|
+
filenameMatchedFiles.push(file);
|
|
383
|
+
}
|
|
384
|
+
try {
|
|
385
|
+
const filePath = join(variantPath, file);
|
|
386
|
+
const content = await readFile(filePath, "utf-8");
|
|
387
|
+
readableFiles++;
|
|
388
|
+
const contentLower = content.toLowerCase();
|
|
389
|
+
let fileMatchCount = 0;
|
|
390
|
+
for (const keyword of keywords) {
|
|
391
|
+
const normalizedKeyword = keyword.toLowerCase();
|
|
392
|
+
if (contentLower.includes(normalizedKeyword)) {
|
|
393
|
+
fileMatchCount++;
|
|
394
|
+
matchedKeywords.add(normalizedKeyword);
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
if (fileMatchCount > 0) {
|
|
398
|
+
matchedFiles.push(file);
|
|
399
|
+
}
|
|
400
|
+
} catch (_error) {
|
|
401
|
+
featureFailedReads.push(file);
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
if (featureFailedReads.length > 0) {
|
|
405
|
+
const totalFiles = Math.min(changedFiles.length, 10);
|
|
406
|
+
console.warn(
|
|
407
|
+
`[requirement-validator] ${featureFailedReads.length}/${totalFiles} files unreadable during feature keyword matching (${featureFailedReads.slice(0, 3).join(", ")})`
|
|
408
|
+
);
|
|
409
|
+
}
|
|
410
|
+
const satisfactionThreshold = Math.min(6, Math.max(1, Math.ceil(keywords.length * 0.3)));
|
|
411
|
+
const matchCount = matchedKeywords.size;
|
|
412
|
+
const satisfied = matchCount >= satisfactionThreshold;
|
|
413
|
+
if (satisfied) {
|
|
414
|
+
return {
|
|
415
|
+
satisfied: true,
|
|
416
|
+
reason: `Feature keywords found in ${matchedFiles.length} files (${matchCount}/${keywords.length} unique keywords)`
|
|
417
|
+
};
|
|
418
|
+
}
|
|
419
|
+
if (readableFiles === 0 && filenameMatchCount > 0) {
|
|
420
|
+
return {
|
|
421
|
+
satisfied: true,
|
|
422
|
+
reason: `Requirement keywords not found in file contents (unreadable), but matched ${filenameMatchCount} filenames: ${filenameMatchedFiles.slice(0, 3).join(", ")}${filenameMatchedFiles.length > 3 ? "..." : ""}`
|
|
423
|
+
};
|
|
424
|
+
}
|
|
425
|
+
return {
|
|
426
|
+
satisfied: false,
|
|
427
|
+
reason: `Insufficient keyword matches: ${matchCount}/${keywords.length} unique keywords (threshold: ${satisfactionThreshold}, readableFiles: ${readableFiles}, filenameMatches: ${filenameMatchCount})`
|
|
428
|
+
};
|
|
429
|
+
}
|
|
430
|
+
function extractKeywords(description) {
|
|
431
|
+
const stopWords = /* @__PURE__ */ new Set([
|
|
432
|
+
"a",
|
|
433
|
+
"an",
|
|
434
|
+
"and",
|
|
435
|
+
"the",
|
|
436
|
+
"to",
|
|
437
|
+
"for",
|
|
438
|
+
"of",
|
|
439
|
+
"in",
|
|
440
|
+
"on",
|
|
441
|
+
"at",
|
|
442
|
+
"by",
|
|
443
|
+
"with",
|
|
444
|
+
"is",
|
|
445
|
+
"are",
|
|
446
|
+
"was",
|
|
447
|
+
"were",
|
|
448
|
+
"be",
|
|
449
|
+
"been",
|
|
450
|
+
"being",
|
|
451
|
+
"have",
|
|
452
|
+
"has",
|
|
453
|
+
"had",
|
|
454
|
+
"do",
|
|
455
|
+
"does",
|
|
456
|
+
"did",
|
|
457
|
+
"will",
|
|
458
|
+
"would",
|
|
459
|
+
"should",
|
|
460
|
+
"could",
|
|
461
|
+
"may",
|
|
462
|
+
"might",
|
|
463
|
+
"must",
|
|
464
|
+
"can",
|
|
465
|
+
"this",
|
|
466
|
+
"that",
|
|
467
|
+
"these",
|
|
468
|
+
"those",
|
|
469
|
+
// Generic instruction words (not useful for verification)
|
|
470
|
+
"something",
|
|
471
|
+
"anything",
|
|
472
|
+
"everything",
|
|
473
|
+
"feature",
|
|
474
|
+
"features",
|
|
475
|
+
"task",
|
|
476
|
+
"change",
|
|
477
|
+
"changes",
|
|
478
|
+
"update",
|
|
479
|
+
"updates",
|
|
480
|
+
"fix",
|
|
481
|
+
"fixes",
|
|
482
|
+
"implement",
|
|
483
|
+
"implements",
|
|
484
|
+
"implementation",
|
|
485
|
+
"add",
|
|
486
|
+
"adds",
|
|
487
|
+
"remove",
|
|
488
|
+
"removes",
|
|
489
|
+
"create",
|
|
490
|
+
"creates",
|
|
491
|
+
"support",
|
|
492
|
+
"supports",
|
|
493
|
+
"ensure",
|
|
494
|
+
"ensures",
|
|
495
|
+
"verify",
|
|
496
|
+
"verifies",
|
|
497
|
+
"validate",
|
|
498
|
+
"validates",
|
|
499
|
+
"tests",
|
|
500
|
+
"test"
|
|
501
|
+
]);
|
|
502
|
+
const tokens = /* @__PURE__ */ new Set();
|
|
503
|
+
const raw = description ?? "";
|
|
504
|
+
const candidates = raw.match(/[A-Za-z][A-Za-z0-9_-]*/g) ?? [];
|
|
505
|
+
const splitCamelCase = (value) => {
|
|
506
|
+
const parts = value.split(/(?<=[a-z0-9])(?=[A-Z])/g);
|
|
507
|
+
return parts.length > 1 ? parts : [value];
|
|
508
|
+
};
|
|
509
|
+
for (const candidate of candidates) {
|
|
510
|
+
const normalized = candidate.replace(/[^A-Za-z0-9_-]/g, "");
|
|
511
|
+
if (!normalized) continue;
|
|
512
|
+
const pieces = normalized.split(/[-_]/g).flatMap((p) => splitCamelCase(p));
|
|
513
|
+
for (const piece of pieces) {
|
|
514
|
+
const lower = piece.toLowerCase();
|
|
515
|
+
if (lower.length <= 2) continue;
|
|
516
|
+
if (stopWords.has(lower)) continue;
|
|
517
|
+
tokens.add(lower);
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
return Array.from(tokens);
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
// src/core/judge/pre-score-gates.ts
|
|
524
|
+
function channelCommands(channel) {
|
|
525
|
+
if (Array.isArray(channel.commands) && channel.commands.length > 0) return channel.commands;
|
|
526
|
+
if (typeof channel.command === "string" && channel.command.trim()) return [channel.command];
|
|
527
|
+
return [];
|
|
528
|
+
}
|
|
529
|
+
var BUILD_CHANNEL_DEFAULT_TIMEOUT_MS = 6e4;
|
|
530
|
+
function classifyBuildGateDominance(result) {
|
|
531
|
+
if (result.exitCode === 124) return "timeout";
|
|
532
|
+
if (result.verificationCoverage?.status === "skipped_no_command" || result.verificationCoverage?.status === "skipped_tool_unavailable") {
|
|
533
|
+
return "unavailable";
|
|
534
|
+
}
|
|
535
|
+
if (result.signal === "fail") return "fail";
|
|
536
|
+
if (result.signal === "pass") return "pass";
|
|
537
|
+
return "unavailable";
|
|
538
|
+
}
|
|
539
|
+
function aggregateBuildGateResults(results) {
|
|
540
|
+
if (results.length === 0) {
|
|
541
|
+
return {
|
|
542
|
+
signal: "unknown",
|
|
543
|
+
errors: [],
|
|
544
|
+
baselineComparisonAvailable: false,
|
|
545
|
+
baselineHealthy: false,
|
|
546
|
+
fatalErrorCount: 0,
|
|
547
|
+
typeErrorCount: 0,
|
|
548
|
+
warningCount: 0,
|
|
549
|
+
verificationCoverage: {
|
|
550
|
+
status: "skipped_no_command",
|
|
551
|
+
reason: "No build commands ran."
|
|
552
|
+
}
|
|
553
|
+
};
|
|
554
|
+
}
|
|
555
|
+
if (results.length === 1) return results[0];
|
|
556
|
+
const representative = results.find((r) => classifyBuildGateDominance(r) === "fail") ?? results.find((r) => classifyBuildGateDominance(r) === "timeout") ?? results.find((r) => classifyBuildGateDominance(r) === "unavailable") ?? results[0];
|
|
557
|
+
const signal = representative.signal;
|
|
558
|
+
const strongerThanUnavailable = results.some((r) => {
|
|
559
|
+
const tier = classifyBuildGateDominance(r);
|
|
560
|
+
return tier === "fail" || tier === "timeout";
|
|
561
|
+
});
|
|
562
|
+
const unavailableCoverage = results.find(
|
|
563
|
+
(r) => r.verificationCoverage?.status === "skipped_no_command" || r.verificationCoverage?.status === "skipped_tool_unavailable"
|
|
564
|
+
)?.verificationCoverage;
|
|
565
|
+
const errors = results.flatMap((r) => r.errors);
|
|
566
|
+
const newErrors = results.some((r) => r.newErrors) ? results.flatMap((r) => r.newErrors ?? []) : void 0;
|
|
567
|
+
return {
|
|
568
|
+
signal,
|
|
569
|
+
errors,
|
|
570
|
+
...newErrors ? {
|
|
571
|
+
newErrors,
|
|
572
|
+
newErrorCount: newErrors.length
|
|
573
|
+
} : {},
|
|
574
|
+
baselineComparisonAvailable: results.some((r) => r.baselineComparisonAvailable),
|
|
575
|
+
baselineHealthy: representative.baselineHealthy,
|
|
576
|
+
fatalErrorCount: results.reduce((sum, r) => sum + r.fatalErrorCount, 0),
|
|
577
|
+
typeErrorCount: results.reduce((sum, r) => sum + r.typeErrorCount, 0),
|
|
578
|
+
warningCount: results.reduce((sum, r) => sum + r.warningCount, 0),
|
|
579
|
+
// Counts are only fully available when every command produced structured counts.
|
|
580
|
+
compilerCountsAvailable: results.every((r) => r.compilerCountsAvailable !== false) ? results.some((r) => r.compilerCountsAvailable === true) ? true : void 0 : false,
|
|
581
|
+
exitCode: representative.exitCode,
|
|
582
|
+
durationMs: results.reduce((sum, r) => sum + (r.durationMs ?? 0), 0),
|
|
583
|
+
command: representative.command,
|
|
584
|
+
stdoutTail: representative.stdoutTail,
|
|
585
|
+
stderrTail: representative.stderrTail,
|
|
586
|
+
outputTail: representative.outputTail,
|
|
587
|
+
verificationCoverage: signal === "pass" ? { status: "verified", command: representative.command } : !strongerThanUnavailable && unavailableCoverage ? unavailableCoverage : signal === "unknown" ? representative.verificationCoverage ?? {
|
|
588
|
+
status: "unverified",
|
|
589
|
+
command: representative.command,
|
|
590
|
+
reason: "One or more scoped build commands could not certify the gate."
|
|
591
|
+
} : representative.verificationCoverage ?? {
|
|
592
|
+
status: "unverified",
|
|
593
|
+
command: representative.command
|
|
594
|
+
}
|
|
595
|
+
};
|
|
596
|
+
}
|
|
597
|
+
function aggregateLintGateResults(artifacts) {
|
|
598
|
+
if (artifacts.length === 0) {
|
|
599
|
+
return {
|
|
600
|
+
passed: false,
|
|
601
|
+
errorCount: 0,
|
|
602
|
+
warningCount: 0,
|
|
603
|
+
filesWithIssues: [],
|
|
604
|
+
exitCode: 1,
|
|
605
|
+
command: "",
|
|
606
|
+
durationMs: 0,
|
|
607
|
+
verificationCoverage: {
|
|
608
|
+
status: "skipped_no_command",
|
|
609
|
+
reason: "No lint commands ran."
|
|
610
|
+
}
|
|
611
|
+
};
|
|
612
|
+
}
|
|
613
|
+
const files = /* @__PURE__ */ new Set();
|
|
614
|
+
let errorCount = 0;
|
|
615
|
+
let warningCount = 0;
|
|
616
|
+
let durationMs = 0;
|
|
617
|
+
for (const artifact of artifacts) {
|
|
618
|
+
errorCount += artifact.result?.errorCount ?? 0;
|
|
619
|
+
warningCount += artifact.result?.warningCount ?? 0;
|
|
620
|
+
durationMs += artifact.durationMs ?? 0;
|
|
621
|
+
for (const file of artifact.result?.filesWithIssues ?? []) {
|
|
622
|
+
files.add(file);
|
|
623
|
+
}
|
|
624
|
+
}
|
|
625
|
+
const anyTimeout = artifacts.some((a) => a.exitCode === 124);
|
|
626
|
+
const anyFailed = artifacts.some(
|
|
627
|
+
(a) => a.exitCode !== 0 && a.exitCode !== 124 && a.verificationCoverage?.status !== "skipped_no_command" && a.verificationCoverage?.status !== "skipped_tool_unavailable"
|
|
628
|
+
);
|
|
629
|
+
const anyUnavailable = artifacts.some(
|
|
630
|
+
(a) => a.verificationCoverage?.status === "skipped_no_command" || a.verificationCoverage?.status === "skipped_tool_unavailable"
|
|
631
|
+
);
|
|
632
|
+
const representative = artifacts.find(
|
|
633
|
+
(a) => a.exitCode !== 0 && a.exitCode !== 124 && a.verificationCoverage?.status !== "skipped_no_command" && a.verificationCoverage?.status !== "skipped_tool_unavailable"
|
|
634
|
+
) ?? artifacts.find((a) => a.exitCode === 124) ?? artifacts.find(
|
|
635
|
+
(a) => a.verificationCoverage?.status === "skipped_no_command" || a.verificationCoverage?.status === "skipped_tool_unavailable"
|
|
636
|
+
) ?? artifacts[0];
|
|
637
|
+
const passed = !anyFailed && !anyTimeout && !anyUnavailable && artifacts.every(lintArtifactPassed);
|
|
638
|
+
return {
|
|
639
|
+
passed,
|
|
640
|
+
errorCount,
|
|
641
|
+
warningCount,
|
|
642
|
+
filesWithIssues: Array.from(files),
|
|
643
|
+
// Dominant representative supplies exit/status — do not force 124 when a
|
|
644
|
+
// real non-timeout failure outranks a concurrent timeout (A8).
|
|
645
|
+
exitCode: representative.exitCode,
|
|
646
|
+
command: representative.command,
|
|
647
|
+
durationMs,
|
|
648
|
+
verificationCoverage: anyFailed || anyTimeout ? representative.verificationCoverage ?? {
|
|
649
|
+
status: "unverified",
|
|
650
|
+
command: representative.command
|
|
651
|
+
} : anyUnavailable ? {
|
|
652
|
+
status: "skipped_tool_unavailable",
|
|
653
|
+
command: representative.command,
|
|
654
|
+
reason: "One or more scoped lint commands were unavailable."
|
|
655
|
+
} : { status: "verified", command: representative.command }
|
|
656
|
+
};
|
|
657
|
+
}
|
|
658
|
+
async function runMultiCommandBuildGate(variantPath, changedFiles, baselineHealthy, commands, options) {
|
|
659
|
+
let remaining = options?.timeoutMs ?? BUILD_CHANNEL_DEFAULT_TIMEOUT_MS;
|
|
660
|
+
const results = [];
|
|
661
|
+
for (const command of commands) {
|
|
662
|
+
if (remaining <= 0) {
|
|
663
|
+
results.push({
|
|
664
|
+
signal: "fail",
|
|
665
|
+
errors: [],
|
|
666
|
+
baselineComparisonAvailable: options?.baselineErrorKeys !== void 0,
|
|
667
|
+
baselineHealthy,
|
|
668
|
+
fatalErrorCount: 0,
|
|
669
|
+
typeErrorCount: 0,
|
|
670
|
+
warningCount: 0,
|
|
671
|
+
compilerCountsAvailable: false,
|
|
672
|
+
exitCode: 124,
|
|
673
|
+
durationMs: 0,
|
|
674
|
+
command,
|
|
675
|
+
outputTail: "Build command skipped: channel timeout budget exhausted.",
|
|
676
|
+
verificationCoverage: {
|
|
677
|
+
status: "unverified",
|
|
678
|
+
command,
|
|
679
|
+
reason: "Channel timeout budget exhausted before this build command ran."
|
|
680
|
+
}
|
|
681
|
+
});
|
|
682
|
+
continue;
|
|
683
|
+
}
|
|
684
|
+
const result = await runBuildGate(variantPath, changedFiles, baselineHealthy, {
|
|
685
|
+
baselineErrorKeys: options?.baselineErrorKeys,
|
|
686
|
+
typecheckExecutable: options?.typecheckExecutable,
|
|
687
|
+
abortSignal: options?.abortSignal,
|
|
688
|
+
command,
|
|
689
|
+
timeoutMs: remaining
|
|
690
|
+
});
|
|
691
|
+
remaining = Math.max(0, remaining - (result.durationMs ?? 0));
|
|
692
|
+
results.push(result);
|
|
693
|
+
}
|
|
694
|
+
return aggregateBuildGateResults(results);
|
|
695
|
+
}
|
|
696
|
+
async function runMultiCommandLintGate(variantPath, commands, options) {
|
|
697
|
+
let remaining = options?.timeoutMs ?? resolveLintGateTimeoutMs();
|
|
698
|
+
const artifacts = [];
|
|
699
|
+
for (const command of commands) {
|
|
700
|
+
if (remaining <= 0) {
|
|
701
|
+
artifacts.push({
|
|
702
|
+
command,
|
|
703
|
+
exitCode: 124,
|
|
704
|
+
durationMs: 0,
|
|
705
|
+
outputTail: "Lint command skipped: channel timeout budget exhausted.",
|
|
706
|
+
result: null,
|
|
707
|
+
verificationCoverage: {
|
|
708
|
+
status: "unverified",
|
|
709
|
+
command,
|
|
710
|
+
reason: "Channel timeout budget exhausted before this lint command ran."
|
|
711
|
+
}
|
|
712
|
+
});
|
|
713
|
+
continue;
|
|
714
|
+
}
|
|
715
|
+
const artifact = await runLintOnDiffWithArtifacts(variantPath, {
|
|
716
|
+
timeoutMs: remaining,
|
|
717
|
+
abortSignal: options?.abortSignal,
|
|
718
|
+
command
|
|
719
|
+
});
|
|
720
|
+
remaining = Math.max(0, remaining - (artifact.durationMs ?? 0));
|
|
721
|
+
artifacts.push(artifact);
|
|
722
|
+
}
|
|
723
|
+
return aggregateLintGateResults(artifacts);
|
|
724
|
+
}
|
|
725
|
+
var NON_SOURCE_EXTENSIONS = /* @__PURE__ */ new Set([
|
|
726
|
+
".md",
|
|
727
|
+
".txt",
|
|
728
|
+
".yaml",
|
|
729
|
+
".yml",
|
|
730
|
+
".toml",
|
|
731
|
+
".csv",
|
|
732
|
+
".xml",
|
|
733
|
+
".svg",
|
|
734
|
+
".png",
|
|
735
|
+
".jpg",
|
|
736
|
+
".jpeg",
|
|
737
|
+
".gif",
|
|
738
|
+
".ico",
|
|
739
|
+
".webp",
|
|
740
|
+
".lock"
|
|
741
|
+
]);
|
|
742
|
+
var NON_SOURCE_BASENAMES = /* @__PURE__ */ new Set([
|
|
743
|
+
".gitignore",
|
|
744
|
+
".editorconfig",
|
|
745
|
+
".prettierrc",
|
|
746
|
+
".eslintignore",
|
|
747
|
+
".npmignore",
|
|
748
|
+
".dockerignore",
|
|
749
|
+
"package-lock.json",
|
|
750
|
+
"pnpm-lock.yaml",
|
|
751
|
+
"yarn.lock",
|
|
752
|
+
"bun.lockb",
|
|
753
|
+
"bun.lock",
|
|
754
|
+
"cargo.lock",
|
|
755
|
+
"gemfile.lock",
|
|
756
|
+
"composer.lock"
|
|
757
|
+
]);
|
|
758
|
+
function isNonSourceFile(filePath) {
|
|
759
|
+
const normalized = path.basename(filePath).toLowerCase();
|
|
760
|
+
if (isProjectConfigPath(filePath)) {
|
|
761
|
+
return false;
|
|
762
|
+
}
|
|
763
|
+
if (NON_SOURCE_BASENAMES.has(normalized)) {
|
|
764
|
+
return true;
|
|
765
|
+
}
|
|
766
|
+
const ext = path.extname(normalized);
|
|
767
|
+
return NON_SOURCE_EXTENSIONS.has(ext);
|
|
768
|
+
}
|
|
769
|
+
function isNonSourceOnly(changedFiles) {
|
|
770
|
+
if (changedFiles.length === 0) return false;
|
|
771
|
+
return changedFiles.every((filePath) => isNonSourceFile(filePath));
|
|
772
|
+
}
|
|
773
|
+
var LINT_GATE_DEFAULT_TIMEOUT_MS = 3e4;
|
|
774
|
+
function resolveLintGateTimeoutMs() {
|
|
775
|
+
const raw = Number(process.env.POETIC_LINT_GATE_TIMEOUT_MS);
|
|
776
|
+
if (!Number.isFinite(raw) || raw <= 0) return LINT_GATE_DEFAULT_TIMEOUT_MS;
|
|
777
|
+
return Math.min(Math.max(raw, 5e3), 12e4);
|
|
778
|
+
}
|
|
779
|
+
function buildTestVerificationCoverage(testGate) {
|
|
780
|
+
if (testGate.availability === "unavailable") {
|
|
781
|
+
return {
|
|
782
|
+
status: "skipped_no_command",
|
|
783
|
+
command: testGate.command,
|
|
784
|
+
reason: testGate.outputTail ?? "Verification unavailable: no task-relevant covering command for the changed files."
|
|
785
|
+
};
|
|
786
|
+
}
|
|
787
|
+
if (testGate.testStatus === "not_run") {
|
|
788
|
+
return {
|
|
789
|
+
status: "skipped_no_command",
|
|
790
|
+
command: testGate.command,
|
|
791
|
+
reason: "Test gate did not run tests."
|
|
792
|
+
};
|
|
793
|
+
}
|
|
794
|
+
if (testGate.testStatus === "timeout" || testGate.testStatus === "error") {
|
|
795
|
+
return {
|
|
796
|
+
status: "unverified",
|
|
797
|
+
command: testGate.command,
|
|
798
|
+
reason: `Test gate ended with status ${testGate.testStatus}.`
|
|
799
|
+
};
|
|
800
|
+
}
|
|
801
|
+
return {
|
|
802
|
+
status: "verified",
|
|
803
|
+
command: testGate.command
|
|
804
|
+
};
|
|
805
|
+
}
|
|
806
|
+
var SCOPED_CHANNEL_UNAVAILABLE_REASON = "Verification unavailable: no task-relevant covering command for the changed files.";
|
|
807
|
+
var SCOPED_CHANNEL_OUT_OF_SCOPE_REASON = "Build channel is out of scope for the changed files.";
|
|
808
|
+
function outOfScopeScopedBuildGate(baselineHealthy) {
|
|
809
|
+
return {
|
|
810
|
+
signal: "unknown",
|
|
811
|
+
errors: [],
|
|
812
|
+
baselineComparisonAvailable: false,
|
|
813
|
+
baselineHealthy,
|
|
814
|
+
fatalErrorCount: 0,
|
|
815
|
+
typeErrorCount: 0,
|
|
816
|
+
warningCount: 0,
|
|
817
|
+
verificationCoverage: {
|
|
818
|
+
status: "skipped_not_relevant",
|
|
819
|
+
reason: SCOPED_CHANNEL_OUT_OF_SCOPE_REASON
|
|
820
|
+
}
|
|
821
|
+
};
|
|
822
|
+
}
|
|
823
|
+
function unavailableScopedBuildGate(baselineHealthy) {
|
|
824
|
+
return {
|
|
825
|
+
signal: "unknown",
|
|
826
|
+
errors: [],
|
|
827
|
+
baselineComparisonAvailable: false,
|
|
828
|
+
baselineHealthy,
|
|
829
|
+
fatalErrorCount: 0,
|
|
830
|
+
typeErrorCount: 0,
|
|
831
|
+
warningCount: 0,
|
|
832
|
+
verificationCoverage: {
|
|
833
|
+
status: "skipped_no_command",
|
|
834
|
+
reason: SCOPED_CHANNEL_UNAVAILABLE_REASON
|
|
835
|
+
}
|
|
836
|
+
};
|
|
837
|
+
}
|
|
838
|
+
function scopedNonRelevantBuildGate(baselineHealthy, availability) {
|
|
839
|
+
if (availability === "out_of_scope") {
|
|
840
|
+
return outOfScopeScopedBuildGate(baselineHealthy);
|
|
841
|
+
}
|
|
842
|
+
return unavailableScopedBuildGate(baselineHealthy);
|
|
843
|
+
}
|
|
844
|
+
function unavailableScopedLintGate() {
|
|
845
|
+
return {
|
|
846
|
+
passed: false,
|
|
847
|
+
errorCount: 0,
|
|
848
|
+
warningCount: 0,
|
|
849
|
+
filesWithIssues: [],
|
|
850
|
+
exitCode: 0,
|
|
851
|
+
command: "",
|
|
852
|
+
durationMs: 0,
|
|
853
|
+
verificationCoverage: {
|
|
854
|
+
status: "skipped_no_command",
|
|
855
|
+
reason: SCOPED_CHANNEL_UNAVAILABLE_REASON
|
|
856
|
+
}
|
|
857
|
+
};
|
|
858
|
+
}
|
|
859
|
+
function unavailableScopedTestGate() {
|
|
860
|
+
return {
|
|
861
|
+
testStatus: "not_run",
|
|
862
|
+
availability: "unavailable",
|
|
863
|
+
skipReason: "no-supported-test-runner",
|
|
864
|
+
testsRun: 0,
|
|
865
|
+
testsPassed: 0,
|
|
866
|
+
testsFailed: 0,
|
|
867
|
+
duration: 0,
|
|
868
|
+
filesTested: [],
|
|
869
|
+
command: "",
|
|
870
|
+
scope: "none",
|
|
871
|
+
outputTail: SCOPED_CHANNEL_UNAVAILABLE_REASON
|
|
872
|
+
};
|
|
873
|
+
}
|
|
874
|
+
var CODE_INTENTS = ["implementation", "bugfix", "refactor"];
|
|
875
|
+
function isCodeIntent(kind) {
|
|
876
|
+
return CODE_INTENTS.includes(kind);
|
|
877
|
+
}
|
|
878
|
+
function isUnavailableVerificationCoverage(coverage) {
|
|
879
|
+
return coverage?.status === "skipped_no_command" || coverage?.status === "skipped_tool_unavailable" || coverage?.status === "unverified";
|
|
880
|
+
}
|
|
881
|
+
function lintArtifactPassed(lintArtifact) {
|
|
882
|
+
if (isUnavailableVerificationCoverage(lintArtifact.verificationCoverage)) {
|
|
883
|
+
return false;
|
|
884
|
+
}
|
|
885
|
+
return lintArtifact.exitCode === 0 && lintArtifact.result?.success === true;
|
|
886
|
+
}
|
|
887
|
+
async function runPreScoreGates(variantPath, changedFiles, requirements, baselineHealthy, options) {
|
|
888
|
+
const runBuildGateEnabled = options?.runBuildGate !== false;
|
|
889
|
+
const runLintGateEnabled = options?.runLintGate === true;
|
|
890
|
+
const variant = options?.variant;
|
|
891
|
+
const shouldRunTestGate = options?.runTestGate ?? false;
|
|
892
|
+
const explicitTestRequirement = hasExplicitTestExecutionRequirement(requirements);
|
|
893
|
+
const verificationChannelScope = resolveVerificationChannelScope(variantPath, changedFiles);
|
|
894
|
+
const channelScopeActive = verificationChannelScope.scoped;
|
|
895
|
+
const buildChannelRelevant = !channelScopeActive || verificationChannelScope.build.relevant === true;
|
|
896
|
+
const lintChannelRelevant = !channelScopeActive || verificationChannelScope.lint.relevant === true;
|
|
897
|
+
const testChannelRelevant = !channelScopeActive || verificationChannelScope.test.relevant === true;
|
|
898
|
+
const scopedBuildCommands = channelScopeActive ? channelCommands(verificationChannelScope.build) : [];
|
|
899
|
+
const scopedLintCommands = channelScopeActive ? channelCommands(verificationChannelScope.lint) : [];
|
|
900
|
+
const scopedBuildCommand = scopedBuildCommands[0];
|
|
901
|
+
const scopedLintCommand = scopedLintCommands[0];
|
|
902
|
+
if (isNonSourceOnly(changedFiles)) {
|
|
903
|
+
const requirementGate2 = await validateRequirements(requirements, variantPath, changedFiles);
|
|
904
|
+
let testGate2;
|
|
905
|
+
if (variant && shouldRunTestGate && explicitTestRequirement) {
|
|
906
|
+
testGate2 = await runTestGate(variant, variantPath, {
|
|
907
|
+
enabled: true,
|
|
908
|
+
mode: options?.testGateMode ?? "touched",
|
|
909
|
+
abortSignal: options?.abortSignal
|
|
910
|
+
});
|
|
911
|
+
}
|
|
912
|
+
const buildGate2 = {
|
|
913
|
+
signal: "unknown",
|
|
914
|
+
errors: [],
|
|
915
|
+
baselineComparisonAvailable: false,
|
|
916
|
+
baselineHealthy,
|
|
917
|
+
fatalErrorCount: 0,
|
|
918
|
+
typeErrorCount: 0,
|
|
919
|
+
warningCount: 0,
|
|
920
|
+
verificationCoverage: {
|
|
921
|
+
status: "skipped_not_relevant",
|
|
922
|
+
reason: "Only non-source files changed."
|
|
923
|
+
}
|
|
924
|
+
};
|
|
925
|
+
const isEligible2 = true;
|
|
926
|
+
const diagnosticReason2 = explicitTestRequirement && (!testGate2 || testGate2.testStatus === "not_run") ? "telemetry: explicit test requirement not met: no tests run" : void 0;
|
|
927
|
+
const buildPassed2 = buildGate2.signal === "pass";
|
|
928
|
+
return {
|
|
929
|
+
requirementGate: requirementGate2,
|
|
930
|
+
buildGate: buildGate2,
|
|
931
|
+
lintGate: void 0,
|
|
932
|
+
testGate: testGate2,
|
|
933
|
+
buildPassed: buildPassed2,
|
|
934
|
+
lintPassed: false,
|
|
935
|
+
isEligible: isEligible2,
|
|
936
|
+
diagnosticReason: diagnosticReason2,
|
|
937
|
+
verificationCoverage: {
|
|
938
|
+
build: buildGate2.verificationCoverage ?? {
|
|
939
|
+
status: "skipped_not_relevant",
|
|
940
|
+
reason: "Only non-source files changed."
|
|
941
|
+
},
|
|
942
|
+
lint: {
|
|
943
|
+
status: "skipped_not_relevant",
|
|
944
|
+
reason: "Only non-source files changed."
|
|
945
|
+
},
|
|
946
|
+
...testGate2 ? {
|
|
947
|
+
test: buildTestVerificationCoverage(testGate2)
|
|
948
|
+
} : {}
|
|
949
|
+
}
|
|
950
|
+
};
|
|
951
|
+
}
|
|
952
|
+
if (!runBuildGateEnabled) {
|
|
953
|
+
const requirementGate2 = await validateRequirements(requirements, variantPath, changedFiles);
|
|
954
|
+
const buildGate2 = {
|
|
955
|
+
signal: "unknown",
|
|
956
|
+
errors: [],
|
|
957
|
+
baselineComparisonAvailable: false,
|
|
958
|
+
baselineHealthy,
|
|
959
|
+
fatalErrorCount: 0,
|
|
960
|
+
typeErrorCount: 0,
|
|
961
|
+
warningCount: 0,
|
|
962
|
+
verificationCoverage: {
|
|
963
|
+
status: "skipped_not_relevant",
|
|
964
|
+
reason: "Build gate was disabled for this run."
|
|
965
|
+
}
|
|
966
|
+
};
|
|
967
|
+
let lintGate2;
|
|
968
|
+
if (runLintGateEnabled) {
|
|
969
|
+
if (!lintChannelRelevant) {
|
|
970
|
+
lintGate2 = unavailableScopedLintGate();
|
|
971
|
+
} else if (scopedLintCommands.length > 0) {
|
|
972
|
+
lintGate2 = await runMultiCommandLintGate(variantPath, scopedLintCommands, {
|
|
973
|
+
abortSignal: options?.abortSignal,
|
|
974
|
+
timeoutMs: resolveLintGateTimeoutMs()
|
|
975
|
+
});
|
|
976
|
+
} else {
|
|
977
|
+
const lintArtifact = await runLintOnDiffWithArtifacts(variantPath, {
|
|
978
|
+
timeoutMs: resolveLintGateTimeoutMs(),
|
|
979
|
+
abortSignal: options?.abortSignal,
|
|
980
|
+
command: scopedLintCommand
|
|
981
|
+
});
|
|
982
|
+
lintGate2 = {
|
|
983
|
+
passed: lintArtifactPassed(lintArtifact),
|
|
984
|
+
errorCount: lintArtifact.result?.errorCount ?? 0,
|
|
985
|
+
warningCount: lintArtifact.result?.warningCount ?? 0,
|
|
986
|
+
filesWithIssues: lintArtifact.result?.filesWithIssues ?? [],
|
|
987
|
+
exitCode: lintArtifact.exitCode,
|
|
988
|
+
command: lintArtifact.command,
|
|
989
|
+
durationMs: lintArtifact.durationMs,
|
|
990
|
+
verificationCoverage: lintArtifact.verificationCoverage
|
|
991
|
+
};
|
|
992
|
+
}
|
|
993
|
+
}
|
|
994
|
+
let testGate2;
|
|
995
|
+
if (variant && shouldRunTestGate) {
|
|
996
|
+
if (!testChannelRelevant) {
|
|
997
|
+
testGate2 = unavailableScopedTestGate();
|
|
998
|
+
} else {
|
|
999
|
+
testGate2 = await runTestGate(variant, variantPath, {
|
|
1000
|
+
enabled: true,
|
|
1001
|
+
mode: options?.testGateMode ?? "touched",
|
|
1002
|
+
abortSignal: options?.abortSignal
|
|
1003
|
+
});
|
|
1004
|
+
}
|
|
1005
|
+
}
|
|
1006
|
+
const buildPassed2 = buildGate2.signal === "pass";
|
|
1007
|
+
const lintPassed2 = lintGate2?.passed ?? false;
|
|
1008
|
+
if (lintGate2 && lintGate2.exitCode > 1) {
|
|
1009
|
+
console.warn(
|
|
1010
|
+
`[pre-score-gates] Lint tool error (exit ${lintGate2.exitCode}) - treating as unknown`
|
|
1011
|
+
);
|
|
1012
|
+
}
|
|
1013
|
+
const isEligible2 = true;
|
|
1014
|
+
const diagnosticReason2 = explicitTestRequirement && (!testGate2 || testGate2.testStatus === "not_run") ? "telemetry: explicit test requirement not met: no tests run" : void 0;
|
|
1015
|
+
return {
|
|
1016
|
+
requirementGate: requirementGate2,
|
|
1017
|
+
buildGate: buildGate2,
|
|
1018
|
+
lintGate: lintGate2,
|
|
1019
|
+
testGate: testGate2,
|
|
1020
|
+
buildPassed: buildPassed2,
|
|
1021
|
+
lintPassed: lintPassed2,
|
|
1022
|
+
isEligible: isEligible2,
|
|
1023
|
+
diagnosticReason: diagnosticReason2,
|
|
1024
|
+
verificationCoverage: {
|
|
1025
|
+
build: buildGate2.verificationCoverage ?? {
|
|
1026
|
+
status: "skipped_not_relevant",
|
|
1027
|
+
reason: "Build gate was disabled for this run."
|
|
1028
|
+
},
|
|
1029
|
+
lint: lintGate2?.verificationCoverage ?? {
|
|
1030
|
+
status: "skipped_not_relevant",
|
|
1031
|
+
reason: "Lint gate was disabled for this run."
|
|
1032
|
+
},
|
|
1033
|
+
...testGate2 ? {
|
|
1034
|
+
test: buildTestVerificationCoverage(testGate2)
|
|
1035
|
+
} : {}
|
|
1036
|
+
}
|
|
1037
|
+
};
|
|
1038
|
+
}
|
|
1039
|
+
const requirementPromise = validateRequirements(requirements, variantPath, changedFiles);
|
|
1040
|
+
const buildPromise = !buildChannelRelevant ? Promise.resolve(
|
|
1041
|
+
scopedNonRelevantBuildGate(baselineHealthy, verificationChannelScope.build.availability)
|
|
1042
|
+
) : scopedBuildCommands.length > 0 ? runMultiCommandBuildGate(variantPath, changedFiles, baselineHealthy, scopedBuildCommands, {
|
|
1043
|
+
baselineErrorKeys: options?.baselineTypecheck?.errorKeys,
|
|
1044
|
+
typecheckExecutable: options?.baselineTypecheck?.typecheckExecutable,
|
|
1045
|
+
abortSignal: options?.abortSignal,
|
|
1046
|
+
timeoutMs: BUILD_CHANNEL_DEFAULT_TIMEOUT_MS
|
|
1047
|
+
}) : runBuildGate(variantPath, changedFiles, baselineHealthy, {
|
|
1048
|
+
baselineErrorKeys: options?.baselineTypecheck?.errorKeys,
|
|
1049
|
+
typecheckExecutable: options?.baselineTypecheck?.typecheckExecutable,
|
|
1050
|
+
abortSignal: options?.abortSignal,
|
|
1051
|
+
command: scopedBuildCommand
|
|
1052
|
+
});
|
|
1053
|
+
const lintPromise = !runLintGateEnabled ? Promise.resolve(null) : !lintChannelRelevant ? Promise.resolve({ kind: "unavailable", gate: unavailableScopedLintGate() }) : scopedLintCommands.length > 0 ? runMultiCommandLintGate(variantPath, scopedLintCommands, {
|
|
1054
|
+
abortSignal: options?.abortSignal,
|
|
1055
|
+
timeoutMs: resolveLintGateTimeoutMs()
|
|
1056
|
+
}).then((gate) => ({ kind: "aggregated", gate })) : runLintOnDiffWithArtifacts(variantPath, {
|
|
1057
|
+
timeoutMs: resolveLintGateTimeoutMs(),
|
|
1058
|
+
abortSignal: options?.abortSignal,
|
|
1059
|
+
command: scopedLintCommand
|
|
1060
|
+
}).then((artifact) => ({ kind: "artifact", artifact }));
|
|
1061
|
+
const testGatePromise = variant && shouldRunTestGate ? !testChannelRelevant ? Promise.resolve(unavailableScopedTestGate()) : runTestGate(variant, variantPath, {
|
|
1062
|
+
enabled: true,
|
|
1063
|
+
mode: options?.testGateMode ?? "touched",
|
|
1064
|
+
abortSignal: options?.abortSignal
|
|
1065
|
+
}) : void 0;
|
|
1066
|
+
const [requirementGate, buildGate, lintOutcome, testGate] = await Promise.all([
|
|
1067
|
+
requirementPromise,
|
|
1068
|
+
buildPromise,
|
|
1069
|
+
lintPromise,
|
|
1070
|
+
testGatePromise ?? Promise.resolve(void 0)
|
|
1071
|
+
]);
|
|
1072
|
+
let lintGate;
|
|
1073
|
+
if (lintOutcome) {
|
|
1074
|
+
if (lintOutcome.kind === "aggregated" || lintOutcome.kind === "unavailable") {
|
|
1075
|
+
lintGate = lintOutcome.gate;
|
|
1076
|
+
} else {
|
|
1077
|
+
const lintArtifact = lintOutcome.artifact;
|
|
1078
|
+
lintGate = {
|
|
1079
|
+
passed: lintArtifactPassed(lintArtifact),
|
|
1080
|
+
errorCount: lintArtifact.result?.errorCount ?? 0,
|
|
1081
|
+
warningCount: lintArtifact.result?.warningCount ?? 0,
|
|
1082
|
+
filesWithIssues: lintArtifact.result?.filesWithIssues ?? [],
|
|
1083
|
+
exitCode: lintArtifact.exitCode,
|
|
1084
|
+
command: lintArtifact.command,
|
|
1085
|
+
durationMs: lintArtifact.durationMs,
|
|
1086
|
+
verificationCoverage: lintArtifact.verificationCoverage
|
|
1087
|
+
};
|
|
1088
|
+
}
|
|
1089
|
+
}
|
|
1090
|
+
const buildPassed = buildGate.signal === "pass";
|
|
1091
|
+
const lintPassed = lintGate?.passed ?? false;
|
|
1092
|
+
if (lintGate && lintGate.exitCode > 1) {
|
|
1093
|
+
console.warn(
|
|
1094
|
+
`[pre-score-gates] Lint tool error (exit ${lintGate.exitCode}) - treating as unknown`
|
|
1095
|
+
);
|
|
1096
|
+
}
|
|
1097
|
+
const isEligible = true;
|
|
1098
|
+
const diagnosticReason = explicitTestRequirement && (!testGate || testGate.testStatus === "not_run") ? "telemetry: explicit test requirement not met: no tests run" : void 0;
|
|
1099
|
+
return {
|
|
1100
|
+
requirementGate,
|
|
1101
|
+
buildGate,
|
|
1102
|
+
lintGate,
|
|
1103
|
+
testGate,
|
|
1104
|
+
buildPassed,
|
|
1105
|
+
lintPassed,
|
|
1106
|
+
isEligible,
|
|
1107
|
+
diagnosticReason,
|
|
1108
|
+
verificationCoverage: {
|
|
1109
|
+
build: buildGate.verificationCoverage ?? {
|
|
1110
|
+
status: buildGate.signal === "pass" ? "verified" : "partially_verified",
|
|
1111
|
+
command: buildGate.command
|
|
1112
|
+
},
|
|
1113
|
+
lint: lintGate?.verificationCoverage ?? {
|
|
1114
|
+
status: runLintGateEnabled ? "unverified" : "skipped_not_relevant",
|
|
1115
|
+
reason: runLintGateEnabled ? "Lint gate returned no artifact." : "Lint gate was disabled."
|
|
1116
|
+
},
|
|
1117
|
+
...testGate ? {
|
|
1118
|
+
test: buildTestVerificationCoverage(testGate)
|
|
1119
|
+
} : {}
|
|
1120
|
+
}
|
|
1121
|
+
};
|
|
1122
|
+
}
|
|
1123
|
+
function hasExplicitTestExecutionRequirement(requirements) {
|
|
1124
|
+
if (!requirements || requirements.length === 0) {
|
|
1125
|
+
return false;
|
|
1126
|
+
}
|
|
1127
|
+
const rebalanceEnabled = JUDGE_FEATURE_FLAGS.TEST_SIGNAL_REBALANCE === "on";
|
|
1128
|
+
for (const req of requirements) {
|
|
1129
|
+
const text = req.description.toLowerCase();
|
|
1130
|
+
if (!rebalanceEnabled) {
|
|
1131
|
+
const legacyPatterns = [
|
|
1132
|
+
/\b\d+\s+tests?\b/i,
|
|
1133
|
+
// "4 tests", "1 test"
|
|
1134
|
+
/\btest cases?\b/i,
|
|
1135
|
+
// "test case", "test cases"
|
|
1136
|
+
/\btest files?\b/i,
|
|
1137
|
+
// "test file", "test files"
|
|
1138
|
+
/\bwrite tests?\b/i,
|
|
1139
|
+
// "write tests"
|
|
1140
|
+
/\badd tests?\b/i,
|
|
1141
|
+
// "add tests"
|
|
1142
|
+
/\bcreate tests?\b/i,
|
|
1143
|
+
// "create tests"
|
|
1144
|
+
/\bimplement tests?\b/i
|
|
1145
|
+
// "implement tests"
|
|
1146
|
+
];
|
|
1147
|
+
if (legacyPatterns.some((pattern) => pattern.test(text))) {
|
|
1148
|
+
return true;
|
|
1149
|
+
}
|
|
1150
|
+
if (req.metadata?.files && req.metadata.files.length > 0) {
|
|
1151
|
+
const hasTestFile = req.metadata.files.some((file) => isProjectTestPath(file));
|
|
1152
|
+
if (hasTestFile) {
|
|
1153
|
+
return true;
|
|
1154
|
+
}
|
|
1155
|
+
}
|
|
1156
|
+
continue;
|
|
1157
|
+
}
|
|
1158
|
+
if (req.type === "tests" && (req.metadata?.testRequirementMode === "run" || req.metadata?.testRequirementMode === "pass")) {
|
|
1159
|
+
return true;
|
|
1160
|
+
}
|
|
1161
|
+
const strictPatterns = [
|
|
1162
|
+
/\btests?\s+(?:must|should|need(?:s)?|has\s+to|required\s+to)\s+pass\b/i,
|
|
1163
|
+
/\bensure\s+(?:the\s+)?tests?\s+pass\b/i,
|
|
1164
|
+
/\ball\s+tests?\s+pass(?:ing)?\b/i,
|
|
1165
|
+
/\b(run|execute|verify|validate)\s+(?:the\s+)?tests?\b/i,
|
|
1166
|
+
/\btests?\s+(?:must|should|need(?:s)?|has\s+to|required\s+to)\s+run\b/i,
|
|
1167
|
+
/\btest\s+suite\s+(?:must|should|need(?:s)?|has\s+to|required\s+to)\s+pass\b/i,
|
|
1168
|
+
/\btest\s+suite\s+(?:must|should|need(?:s)?|has\s+to|required\s+to)\s+run\b/i
|
|
1169
|
+
];
|
|
1170
|
+
if (strictPatterns.some((pattern) => pattern.test(text))) {
|
|
1171
|
+
return true;
|
|
1172
|
+
}
|
|
1173
|
+
}
|
|
1174
|
+
return false;
|
|
1175
|
+
}
|
|
1176
|
+
|
|
1177
|
+
// src/core/types/percent-score.ts
|
|
1178
|
+
var SCORE_MIN = 0;
|
|
1179
|
+
var SCORE_MAX = 100;
|
|
1180
|
+
function createPercentScore(value, context) {
|
|
1181
|
+
if (!Number.isFinite(value)) {
|
|
1182
|
+
throw new Error(`${context || "PercentScore"} must be a finite number, received ${value}`);
|
|
1183
|
+
}
|
|
1184
|
+
if (value < SCORE_MIN || value > SCORE_MAX) {
|
|
1185
|
+
throw new Error(
|
|
1186
|
+
`${context || "PercentScore"} must be between ${SCORE_MIN} and ${SCORE_MAX}, received ${value}`
|
|
1187
|
+
);
|
|
1188
|
+
}
|
|
1189
|
+
return value;
|
|
1190
|
+
}
|
|
1191
|
+
function percentScoreToNumber(score) {
|
|
1192
|
+
return score;
|
|
1193
|
+
}
|
|
1194
|
+
var PERCENT_SCORES = {
|
|
1195
|
+
ZERO: createPercentScore(0, "constant"),
|
|
1196
|
+
QUARTER: createPercentScore(25, "constant"),
|
|
1197
|
+
HALF: createPercentScore(50, "constant"),
|
|
1198
|
+
THREE_QUARTERS: createPercentScore(75, "constant"),
|
|
1199
|
+
FULL: createPercentScore(100, "constant"),
|
|
1200
|
+
DEFAULT_THRESHOLD: createPercentScore(60, "constant"),
|
|
1201
|
+
HIGH_CONFIDENCE: createPercentScore(80, "constant"),
|
|
1202
|
+
MAXIMUM_CONFIDENCE: createPercentScore(95, "constant")
|
|
1203
|
+
};
|
|
1204
|
+
|
|
1205
|
+
// src/core/judge/structural-risk-scanner.ts
|
|
1206
|
+
var FUNCTION_DECLARATION_PATTERN = /^\s*(?:export\s+)?(?:async\s+)?function\s+([A-Za-z_$][\w$]*)\s*\(/;
|
|
1207
|
+
var METHOD_DECLARATION_PATTERN = /^\s*(?:(?:public|private|protected|static|readonly|override|async|get|set)\s+)*([A-Za-z_$][\w$]*)\s*\([^)]*\)\s*(?::[^{]+)?\s*\{/;
|
|
1208
|
+
var ARROW_FUNCTION_PATTERN = /^\s*(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*(?:async\s*)?\([^)]*\)\s*=>\s*\{/;
|
|
1209
|
+
var METHOD_KEYWORDS = /* @__PURE__ */ new Set(["if", "for", "while", "switch", "catch"]);
|
|
1210
|
+
var STRUCTURAL_RISK_APPLICABLE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
1211
|
+
"typescript",
|
|
1212
|
+
"javascript",
|
|
1213
|
+
"node"
|
|
1214
|
+
]);
|
|
1215
|
+
function isStructuralRiskApplicableLanguage(language) {
|
|
1216
|
+
return language !== null && STRUCTURAL_RISK_APPLICABLE_LANGUAGES.has(language);
|
|
1217
|
+
}
|
|
1218
|
+
function resolveApplicability(applicableCount, unsupportedSourceCount) {
|
|
1219
|
+
if (applicableCount === 0 && unsupportedSourceCount === 0) return "unknown";
|
|
1220
|
+
if (applicableCount > 0 && unsupportedSourceCount === 0) return "applicable";
|
|
1221
|
+
if (applicableCount === 0 && unsupportedSourceCount > 0) return "not_applicable";
|
|
1222
|
+
return "mixed";
|
|
1223
|
+
}
|
|
1224
|
+
function scanDiffForStructuralRisks(diffText) {
|
|
1225
|
+
if (!diffText || diffText.trim().length === 0) {
|
|
1226
|
+
return emptyResult("unknown");
|
|
1227
|
+
}
|
|
1228
|
+
const sections = parseFileSections(diffText);
|
|
1229
|
+
if (sections.length === 0) {
|
|
1230
|
+
return emptyResult("unknown");
|
|
1231
|
+
}
|
|
1232
|
+
const findings = [];
|
|
1233
|
+
const languagesSeen = /* @__PURE__ */ new Set();
|
|
1234
|
+
let applicableCount = 0;
|
|
1235
|
+
let unsupportedSourceCount = 0;
|
|
1236
|
+
for (const section of sections) {
|
|
1237
|
+
const language = languageForProjectPath(section.path);
|
|
1238
|
+
if (language === null || language === "docs" || language === "unknown") {
|
|
1239
|
+
continue;
|
|
1240
|
+
}
|
|
1241
|
+
languagesSeen.add(language);
|
|
1242
|
+
if (!isStructuralRiskApplicableLanguage(language)) {
|
|
1243
|
+
unsupportedSourceCount += 1;
|
|
1244
|
+
continue;
|
|
1245
|
+
}
|
|
1246
|
+
applicableCount += 1;
|
|
1247
|
+
findings.push(...detectSelfRecursiveCalls(section));
|
|
1248
|
+
findings.push(...detectEmptyCatchBlocks(section));
|
|
1249
|
+
findings.push(...detectUnreachableAfterReturn(section));
|
|
1250
|
+
findings.push(...detectSwitchFallthrough(section));
|
|
1251
|
+
}
|
|
1252
|
+
const criticalCount = findings.filter((finding) => finding.severity === "critical").length;
|
|
1253
|
+
const majorCount = findings.filter((finding) => finding.severity === "major").length;
|
|
1254
|
+
const minorCount = findings.filter((finding) => finding.severity === "minor").length;
|
|
1255
|
+
const applicability = resolveApplicability(applicableCount, unsupportedSourceCount);
|
|
1256
|
+
return {
|
|
1257
|
+
findings,
|
|
1258
|
+
criticalCount,
|
|
1259
|
+
majorCount,
|
|
1260
|
+
minorCount,
|
|
1261
|
+
hasCritical: criticalCount > 0,
|
|
1262
|
+
applicability,
|
|
1263
|
+
languagesSeen: [...languagesSeen]
|
|
1264
|
+
};
|
|
1265
|
+
}
|
|
1266
|
+
function emptyResult(applicability = "unknown") {
|
|
1267
|
+
return {
|
|
1268
|
+
findings: [],
|
|
1269
|
+
criticalCount: 0,
|
|
1270
|
+
majorCount: 0,
|
|
1271
|
+
minorCount: 0,
|
|
1272
|
+
hasCritical: false,
|
|
1273
|
+
applicability,
|
|
1274
|
+
languagesSeen: []
|
|
1275
|
+
};
|
|
1276
|
+
}
|
|
1277
|
+
function parseFileSections(diffText) {
|
|
1278
|
+
const lines = diffText.replace(/\r\n/g, "\n").split("\n");
|
|
1279
|
+
const sections = [];
|
|
1280
|
+
let current = null;
|
|
1281
|
+
let currentNewLine = 0;
|
|
1282
|
+
const flush = () => {
|
|
1283
|
+
if (current) {
|
|
1284
|
+
sections.push(current);
|
|
1285
|
+
current = null;
|
|
1286
|
+
}
|
|
1287
|
+
};
|
|
1288
|
+
for (const line of lines) {
|
|
1289
|
+
if (line.startsWith("diff --git ")) {
|
|
1290
|
+
flush();
|
|
1291
|
+
const paths = parseDiffGitHeaderPaths(line);
|
|
1292
|
+
const path3 = paths?.bPath ?? paths?.aPath ?? line.replace(/^diff --git\s+/, "");
|
|
1293
|
+
current = {
|
|
1294
|
+
path: normalizePath(path3),
|
|
1295
|
+
index: sections.length,
|
|
1296
|
+
text: `${line}
|
|
1297
|
+
`,
|
|
1298
|
+
addedLines: []
|
|
1299
|
+
};
|
|
1300
|
+
currentNewLine = 0;
|
|
1301
|
+
continue;
|
|
1302
|
+
}
|
|
1303
|
+
if (!current) continue;
|
|
1304
|
+
current.text += `${line}
|
|
1305
|
+
`;
|
|
1306
|
+
const hunkMatch = line.match(/^@@ -\d+(?:,\d+)? \+(\d+)(?:,\d+)? @@/);
|
|
1307
|
+
if (hunkMatch) {
|
|
1308
|
+
currentNewLine = Number(hunkMatch[1] ?? "0");
|
|
1309
|
+
continue;
|
|
1310
|
+
}
|
|
1311
|
+
if (line.startsWith("+++ ") || line.startsWith("--- ")) continue;
|
|
1312
|
+
if (line.startsWith("+")) {
|
|
1313
|
+
current.addedLines.push({
|
|
1314
|
+
line: currentNewLine > 0 ? currentNewLine : null,
|
|
1315
|
+
text: line.slice(1)
|
|
1316
|
+
});
|
|
1317
|
+
currentNewLine += 1;
|
|
1318
|
+
continue;
|
|
1319
|
+
}
|
|
1320
|
+
if (line.startsWith("-")) {
|
|
1321
|
+
continue;
|
|
1322
|
+
}
|
|
1323
|
+
if (line.startsWith(" ")) {
|
|
1324
|
+
currentNewLine += 1;
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
flush();
|
|
1328
|
+
return sections;
|
|
1329
|
+
}
|
|
1330
|
+
function detectSelfRecursiveCalls(section) {
|
|
1331
|
+
const findings = [];
|
|
1332
|
+
const blocks = extractFunctionBlocks(section);
|
|
1333
|
+
for (const block of blocks) {
|
|
1334
|
+
const nameEscaped = escapeRegex(block.name);
|
|
1335
|
+
const selfCallPattern = new RegExp(
|
|
1336
|
+
`(?<![.\\w$])(?:return\\s+)?(?:this\\.)?${nameEscaped}\\s*\\(`
|
|
1337
|
+
);
|
|
1338
|
+
const selfCallIndex = block.lines.findIndex((line, index) => {
|
|
1339
|
+
if (index === 0) return false;
|
|
1340
|
+
const text = line.text.trim();
|
|
1341
|
+
if (!text || text.startsWith("//")) return false;
|
|
1342
|
+
return selfCallPattern.test(text);
|
|
1343
|
+
});
|
|
1344
|
+
if (selfCallIndex < 0) continue;
|
|
1345
|
+
const hasBaseCase = hasBaseCaseBeforeRecursiveCall(block.lines, selfCallIndex);
|
|
1346
|
+
if (hasBaseCase) continue;
|
|
1347
|
+
const selfCallLine = block.lines[selfCallIndex];
|
|
1348
|
+
findings.push({
|
|
1349
|
+
severity: "critical",
|
|
1350
|
+
kind: "self_recursive_call",
|
|
1351
|
+
summary: `Self-referential call detected in ${block.name}() without clear termination condition`,
|
|
1352
|
+
filePath: block.filePath,
|
|
1353
|
+
line: selfCallLine?.line ?? void 0,
|
|
1354
|
+
evidence: selfCallLine?.text.trim()
|
|
1355
|
+
});
|
|
1356
|
+
}
|
|
1357
|
+
return findings;
|
|
1358
|
+
}
|
|
1359
|
+
function extractFunctionBlocks(section) {
|
|
1360
|
+
const blocks = [];
|
|
1361
|
+
let active;
|
|
1362
|
+
for (const line of section.addedLines) {
|
|
1363
|
+
const text = line.text;
|
|
1364
|
+
const signatureName = detectFunctionName(text);
|
|
1365
|
+
if (!active && signatureName) {
|
|
1366
|
+
const open2 = countChar(text, "{");
|
|
1367
|
+
const close2 = countChar(text, "}");
|
|
1368
|
+
active = {
|
|
1369
|
+
name: signatureName,
|
|
1370
|
+
braceDepth: open2 - close2,
|
|
1371
|
+
sawOpeningBrace: open2 > 0,
|
|
1372
|
+
lines: [line]
|
|
1373
|
+
};
|
|
1374
|
+
if (active.sawOpeningBrace && active.braceDepth <= 0) {
|
|
1375
|
+
blocks.push({
|
|
1376
|
+
name: active.name,
|
|
1377
|
+
filePath: section.path,
|
|
1378
|
+
lines: active.lines
|
|
1379
|
+
});
|
|
1380
|
+
active = void 0;
|
|
1381
|
+
}
|
|
1382
|
+
continue;
|
|
1383
|
+
}
|
|
1384
|
+
if (!active) continue;
|
|
1385
|
+
active.lines.push(line);
|
|
1386
|
+
const open = countChar(text, "{");
|
|
1387
|
+
const close = countChar(text, "}");
|
|
1388
|
+
if (open > 0) active.sawOpeningBrace = true;
|
|
1389
|
+
active.braceDepth += open - close;
|
|
1390
|
+
if (active.sawOpeningBrace && active.braceDepth <= 0) {
|
|
1391
|
+
blocks.push({
|
|
1392
|
+
name: active.name,
|
|
1393
|
+
filePath: section.path,
|
|
1394
|
+
lines: active.lines
|
|
1395
|
+
});
|
|
1396
|
+
active = void 0;
|
|
1397
|
+
}
|
|
1398
|
+
}
|
|
1399
|
+
if (active && active.lines.length > 0) {
|
|
1400
|
+
blocks.push({
|
|
1401
|
+
name: active.name,
|
|
1402
|
+
filePath: section.path,
|
|
1403
|
+
lines: active.lines
|
|
1404
|
+
});
|
|
1405
|
+
}
|
|
1406
|
+
return blocks;
|
|
1407
|
+
}
|
|
1408
|
+
function detectFunctionName(line) {
|
|
1409
|
+
const declaration = line.match(FUNCTION_DECLARATION_PATTERN);
|
|
1410
|
+
if (declaration?.[1]) return declaration[1];
|
|
1411
|
+
const arrow = line.match(ARROW_FUNCTION_PATTERN);
|
|
1412
|
+
if (arrow?.[1]) return arrow[1];
|
|
1413
|
+
const method = line.match(METHOD_DECLARATION_PATTERN);
|
|
1414
|
+
if (method?.[1]) {
|
|
1415
|
+
const candidate = method[1];
|
|
1416
|
+
if (!METHOD_KEYWORDS.has(candidate)) {
|
|
1417
|
+
return candidate;
|
|
1418
|
+
}
|
|
1419
|
+
}
|
|
1420
|
+
return null;
|
|
1421
|
+
}
|
|
1422
|
+
function hasBaseCaseBeforeRecursiveCall(lines, recursiveCallIndex) {
|
|
1423
|
+
const prior = lines.slice(0, recursiveCallIndex + 1).map((line) => line.text.trim());
|
|
1424
|
+
for (let i = 0; i < prior.length; i += 1) {
|
|
1425
|
+
const text = prior[i];
|
|
1426
|
+
if (!text) continue;
|
|
1427
|
+
if (/\bif\s*\(/.test(text) && (/\breturn\b/.test(text) || /\bthrow\b/.test(text) || /\bbreak\b/.test(text))) {
|
|
1428
|
+
return true;
|
|
1429
|
+
}
|
|
1430
|
+
if (/\bif\s*\(/.test(text) && i + 1 < prior.length) {
|
|
1431
|
+
const next = prior[i + 1] ?? "";
|
|
1432
|
+
if (/\breturn\b/.test(next) || /\bthrow\b/.test(next)) {
|
|
1433
|
+
return true;
|
|
1434
|
+
}
|
|
1435
|
+
}
|
|
1436
|
+
}
|
|
1437
|
+
return false;
|
|
1438
|
+
}
|
|
1439
|
+
function detectEmptyCatchBlocks(section) {
|
|
1440
|
+
const findings = [];
|
|
1441
|
+
const lines = section.addedLines;
|
|
1442
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
1443
|
+
const current = lines[i];
|
|
1444
|
+
if (!current) continue;
|
|
1445
|
+
const text = current.text;
|
|
1446
|
+
if (/catch\s*\([^)]*\)\s*{\s*}/.test(text)) {
|
|
1447
|
+
findings.push({
|
|
1448
|
+
severity: "major",
|
|
1449
|
+
kind: "empty_catch",
|
|
1450
|
+
summary: "Empty catch block detected",
|
|
1451
|
+
filePath: section.path,
|
|
1452
|
+
line: current.line ?? void 0,
|
|
1453
|
+
evidence: text.trim()
|
|
1454
|
+
});
|
|
1455
|
+
continue;
|
|
1456
|
+
}
|
|
1457
|
+
if (!/catch\s*\([^)]*\)\s*{/.test(text)) continue;
|
|
1458
|
+
let depth = countChar(text, "{") - countChar(text, "}");
|
|
1459
|
+
const blockLines = [text];
|
|
1460
|
+
let j = i + 1;
|
|
1461
|
+
for (; j < lines.length && depth > 0; j += 1) {
|
|
1462
|
+
const candidate = lines[j];
|
|
1463
|
+
if (!candidate) break;
|
|
1464
|
+
blockLines.push(candidate.text);
|
|
1465
|
+
depth += countChar(candidate.text, "{") - countChar(candidate.text, "}");
|
|
1466
|
+
}
|
|
1467
|
+
const meaningful = blockLines.map((line) => line.trim()).filter(
|
|
1468
|
+
(line) => line.length > 0 && line !== "{" && line !== "}" && !line.startsWith("//") && !line.startsWith("/*") && !line.startsWith("*") && !line.startsWith("catch")
|
|
1469
|
+
);
|
|
1470
|
+
if (meaningful.length === 0) {
|
|
1471
|
+
findings.push({
|
|
1472
|
+
severity: "major",
|
|
1473
|
+
kind: "empty_catch",
|
|
1474
|
+
summary: "Catch block swallows errors without handling",
|
|
1475
|
+
filePath: section.path,
|
|
1476
|
+
line: current.line ?? void 0,
|
|
1477
|
+
evidence: current.text.trim()
|
|
1478
|
+
});
|
|
1479
|
+
}
|
|
1480
|
+
i = j - 1;
|
|
1481
|
+
}
|
|
1482
|
+
return findings;
|
|
1483
|
+
}
|
|
1484
|
+
function detectUnreachableAfterReturn(section) {
|
|
1485
|
+
const findings = [];
|
|
1486
|
+
const blocks = extractFunctionBlocks(section);
|
|
1487
|
+
for (const block of blocks) {
|
|
1488
|
+
for (let i = 0; i < block.lines.length - 1; i += 1) {
|
|
1489
|
+
const current = block.lines[i];
|
|
1490
|
+
if (!current) continue;
|
|
1491
|
+
const currentText = current.text.trim();
|
|
1492
|
+
if (!/^(return|throw)\b/.test(currentText)) continue;
|
|
1493
|
+
const currentIndent = indentation(current.text);
|
|
1494
|
+
const next = findNextMeaningfulLine(block.lines, i + 1);
|
|
1495
|
+
if (!next) continue;
|
|
1496
|
+
const nextIndent = indentation(next.text);
|
|
1497
|
+
const nextTrimmed = next.text.trim();
|
|
1498
|
+
if (nextIndent === currentIndent && !nextTrimmed.startsWith("}") && !nextTrimmed.startsWith("else") && !nextTrimmed.startsWith("case ") && !nextTrimmed.startsWith("default:")) {
|
|
1499
|
+
findings.push({
|
|
1500
|
+
severity: "minor",
|
|
1501
|
+
kind: "unreachable_after_return",
|
|
1502
|
+
summary: "Possible unreachable code immediately after unconditional return/throw",
|
|
1503
|
+
filePath: block.filePath,
|
|
1504
|
+
line: next.line ?? void 0,
|
|
1505
|
+
evidence: nextTrimmed
|
|
1506
|
+
});
|
|
1507
|
+
}
|
|
1508
|
+
}
|
|
1509
|
+
}
|
|
1510
|
+
return findings;
|
|
1511
|
+
}
|
|
1512
|
+
function detectSwitchFallthrough(section) {
|
|
1513
|
+
const findings = [];
|
|
1514
|
+
const lines = section.addedLines;
|
|
1515
|
+
let inSwitch = false;
|
|
1516
|
+
let switchDepth = 0;
|
|
1517
|
+
let caseStart = null;
|
|
1518
|
+
let caseBody = [];
|
|
1519
|
+
const flushCase = () => {
|
|
1520
|
+
if (!caseStart) {
|
|
1521
|
+
caseBody = [];
|
|
1522
|
+
return;
|
|
1523
|
+
}
|
|
1524
|
+
const hasExplicitFlowEnd = caseBody.some(
|
|
1525
|
+
(line) => /\b(break|return|throw|continue)\b/.test(line)
|
|
1526
|
+
);
|
|
1527
|
+
const hasFallthroughAnnotation = caseBody.some((line) => /fall.?through/i.test(line));
|
|
1528
|
+
const hasExecutableContent = caseBody.some((line) => {
|
|
1529
|
+
const trimmed = line.trim();
|
|
1530
|
+
return trimmed.length > 0 && !trimmed.startsWith("//") && trimmed !== "{" && trimmed !== "}";
|
|
1531
|
+
});
|
|
1532
|
+
if (!hasExplicitFlowEnd && !hasFallthroughAnnotation && hasExecutableContent) {
|
|
1533
|
+
findings.push({
|
|
1534
|
+
severity: "major",
|
|
1535
|
+
kind: "switch_fallthrough",
|
|
1536
|
+
summary: "Switch case may fall through without break/return/throw",
|
|
1537
|
+
filePath: section.path,
|
|
1538
|
+
line: caseStart.line ?? void 0,
|
|
1539
|
+
evidence: caseStart.text.trim()
|
|
1540
|
+
});
|
|
1541
|
+
}
|
|
1542
|
+
caseStart = null;
|
|
1543
|
+
caseBody = [];
|
|
1544
|
+
};
|
|
1545
|
+
for (const line of lines) {
|
|
1546
|
+
const text = line.text;
|
|
1547
|
+
const trimmed = text.trim();
|
|
1548
|
+
if (!inSwitch && /\bswitch\s*\(/.test(trimmed)) {
|
|
1549
|
+
inSwitch = true;
|
|
1550
|
+
switchDepth = countChar(trimmed, "{") - countChar(trimmed, "}");
|
|
1551
|
+
continue;
|
|
1552
|
+
}
|
|
1553
|
+
if (!inSwitch) continue;
|
|
1554
|
+
switchDepth += countChar(trimmed, "{") - countChar(trimmed, "}");
|
|
1555
|
+
if (switchDepth <= 0) {
|
|
1556
|
+
flushCase();
|
|
1557
|
+
inSwitch = false;
|
|
1558
|
+
switchDepth = 0;
|
|
1559
|
+
continue;
|
|
1560
|
+
}
|
|
1561
|
+
if (/^(case\b.+:|default:)/.test(trimmed)) {
|
|
1562
|
+
flushCase();
|
|
1563
|
+
caseStart = line;
|
|
1564
|
+
continue;
|
|
1565
|
+
}
|
|
1566
|
+
if (caseStart) {
|
|
1567
|
+
caseBody.push(trimmed);
|
|
1568
|
+
}
|
|
1569
|
+
}
|
|
1570
|
+
flushCase();
|
|
1571
|
+
return findings;
|
|
1572
|
+
}
|
|
1573
|
+
function findNextMeaningfulLine(lines, start) {
|
|
1574
|
+
for (let i = start; i < lines.length; i += 1) {
|
|
1575
|
+
const candidate = lines[i];
|
|
1576
|
+
if (!candidate) continue;
|
|
1577
|
+
const trimmed = candidate.text.trim();
|
|
1578
|
+
if (!trimmed || trimmed.startsWith("//")) continue;
|
|
1579
|
+
return candidate;
|
|
1580
|
+
}
|
|
1581
|
+
return null;
|
|
1582
|
+
}
|
|
1583
|
+
function normalizePath(path3) {
|
|
1584
|
+
return path3.replace(/\\/g, "/").replace(/^\.?\//, "").trim();
|
|
1585
|
+
}
|
|
1586
|
+
function countChar(text, char) {
|
|
1587
|
+
let count = 0;
|
|
1588
|
+
for (const c of text) {
|
|
1589
|
+
if (c === char) count += 1;
|
|
1590
|
+
}
|
|
1591
|
+
return count;
|
|
1592
|
+
}
|
|
1593
|
+
function escapeRegex(value) {
|
|
1594
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
1595
|
+
}
|
|
1596
|
+
function indentation(line) {
|
|
1597
|
+
const match = line.match(/^\s*/);
|
|
1598
|
+
return match?.[0].length ?? 0;
|
|
1599
|
+
}
|
|
1600
|
+
|
|
1601
|
+
// src/core/judge/ac-compliance-scanner.ts
|
|
1602
|
+
function escapeRegExp(s) {
|
|
1603
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
1604
|
+
}
|
|
1605
|
+
function extractAddedDiffBody(diffText) {
|
|
1606
|
+
if (!diffText) return "";
|
|
1607
|
+
const lines = diffText.split("\n");
|
|
1608
|
+
let sawUnifiedDiffHeader = false;
|
|
1609
|
+
const addedLines = [];
|
|
1610
|
+
for (const line of lines) {
|
|
1611
|
+
if (line.startsWith("diff --git ")) {
|
|
1612
|
+
sawUnifiedDiffHeader = true;
|
|
1613
|
+
continue;
|
|
1614
|
+
}
|
|
1615
|
+
if (line.startsWith("@@") || line.startsWith("+++ ") || line.startsWith("--- ")) {
|
|
1616
|
+
continue;
|
|
1617
|
+
}
|
|
1618
|
+
if (line.startsWith("+")) {
|
|
1619
|
+
addedLines.push(line.slice(1));
|
|
1620
|
+
}
|
|
1621
|
+
}
|
|
1622
|
+
if (addedLines.length === 0 && !sawUnifiedDiffHeader) {
|
|
1623
|
+
return diffText.trim();
|
|
1624
|
+
}
|
|
1625
|
+
return addedLines.join("\n").trim();
|
|
1626
|
+
}
|
|
1627
|
+
function extractKeySymbols(acText) {
|
|
1628
|
+
const camelCase = acText.match(/\b[a-z][a-zA-Z0-9]*[A-Z][a-zA-Z0-9]*\b/g) ?? [];
|
|
1629
|
+
const pascalCase = acText.match(/\b[A-Z][a-z]+(?:[A-Z][a-z]+)+\b/g) ?? [];
|
|
1630
|
+
const backtickTerms = (acText.match(/`([^`]+)`/g) ?? []).map((s) => s.replace(/`/g, ""));
|
|
1631
|
+
const snakeCase = acText.match(/\b[a-z][a-z0-9]*(?:_[a-z0-9]+)+\b/g) ?? [];
|
|
1632
|
+
const symbolPool = backtickTerms.length > 0 ? backtickTerms : [.../* @__PURE__ */ new Set([...camelCase, ...pascalCase, ...snakeCase])];
|
|
1633
|
+
const combined = [...new Set(symbolPool)];
|
|
1634
|
+
return combined.filter(
|
|
1635
|
+
(s) => s.length >= 4 && !/^(this|that|when|then|from|with|must|have|each|only)$/i.test(s)
|
|
1636
|
+
);
|
|
1637
|
+
}
|
|
1638
|
+
function evaluateACComplianceItems(contract, variant) {
|
|
1639
|
+
if (!contract) return [];
|
|
1640
|
+
const items = [
|
|
1641
|
+
...contract.mustHaves.map((item) => ({ item, source: "MUST" })),
|
|
1642
|
+
...contract.acceptanceCriteria.map((item) => ({ item, source: "AC" }))
|
|
1643
|
+
].filter(({ item }) => item.verification !== "gate");
|
|
1644
|
+
if (items.length === 0) return [];
|
|
1645
|
+
const addedDiffBody = extractAddedDiffBody(extractDiffText(variant));
|
|
1646
|
+
const normalizedAddedDiffBody = addedDiffBody.toLowerCase();
|
|
1647
|
+
const symbolRegexCache = /* @__PURE__ */ new Map();
|
|
1648
|
+
const getSymbolRegex = (sym) => {
|
|
1649
|
+
let regex = symbolRegexCache.get(sym);
|
|
1650
|
+
if (!regex) {
|
|
1651
|
+
regex = new RegExp(`\\b${escapeRegExp(sym)}\\b`, "i");
|
|
1652
|
+
symbolRegexCache.set(sym, regex);
|
|
1653
|
+
}
|
|
1654
|
+
return regex;
|
|
1655
|
+
};
|
|
1656
|
+
const results = [];
|
|
1657
|
+
for (const { item, source } of items) {
|
|
1658
|
+
const extractedSymbols = extractKeySymbols(item.text);
|
|
1659
|
+
const probeTerms = extractedSymbols.length > 0 ? extractedSymbols : [item.text.replace(/\s+/g, " ").trim()];
|
|
1660
|
+
const matchedSymbols = probeTerms.filter((sym) => {
|
|
1661
|
+
if (!sym) return false;
|
|
1662
|
+
if (!normalizedAddedDiffBody) return false;
|
|
1663
|
+
return getSymbolRegex(sym).test(addedDiffBody);
|
|
1664
|
+
});
|
|
1665
|
+
const status = matchedSymbols.length === 0 ? "NOT FOUND" : matchedSymbols.length === probeTerms.length ? "FOUND" : "PARTIAL";
|
|
1666
|
+
results.push({
|
|
1667
|
+
source,
|
|
1668
|
+
itemText: item.text,
|
|
1669
|
+
status,
|
|
1670
|
+
symbols: probeTerms,
|
|
1671
|
+
matchedSymbols
|
|
1672
|
+
});
|
|
1673
|
+
}
|
|
1674
|
+
return results;
|
|
1675
|
+
}
|
|
1676
|
+
function buildACComplianceSection(contract, variant) {
|
|
1677
|
+
const results = evaluateACComplianceItems(contract, variant);
|
|
1678
|
+
if (results.length === 0) return "";
|
|
1679
|
+
const lines = ["**AC Compliance Scan** (automated \u2014 presence/absence only):"];
|
|
1680
|
+
for (const result of results) {
|
|
1681
|
+
lines.push(
|
|
1682
|
+
`- [${result.status}] [${result.source}] ${result.itemText} (scanned for: ${result.symbols.join(", ")}; matched: ${result.matchedSymbols.join(", ") || "none"})`
|
|
1683
|
+
);
|
|
1684
|
+
}
|
|
1685
|
+
return lines.join("\n");
|
|
1686
|
+
}
|
|
1687
|
+
var BUGFIX_KEYWORD_PATTERN = /\b(fix|bug|patch|repair|resolve)\b/i;
|
|
1688
|
+
var PARSING_VALIDATION_SYMBOL_PATTERN = /\b(split|parse|validate|normalize|canonical|isValid|isCanonical|formatVariantIdentifier|parseVariantIdentifier|regex|regexp|matchAll?)\b/i;
|
|
1689
|
+
function variantHasParsingValidationChanges(variant) {
|
|
1690
|
+
const addedDiffBody = extractAddedDiffBody(extractDiffText(variant));
|
|
1691
|
+
if (!addedDiffBody) return false;
|
|
1692
|
+
return PARSING_VALIDATION_SYMBOL_PATTERN.test(addedDiffBody);
|
|
1693
|
+
}
|
|
1694
|
+
function requiresInputSimulationCheck(variants, taskPrompt, taskKind) {
|
|
1695
|
+
const isBugfix = taskKind === "bugfix" || BUGFIX_KEYWORD_PATTERN.test(taskPrompt);
|
|
1696
|
+
if (!isBugfix) return false;
|
|
1697
|
+
return variants.some((variant) => variantHasParsingValidationChanges(variant));
|
|
1698
|
+
}
|
|
1699
|
+
|
|
1700
|
+
// src/core/judge/ai-judge-prompt-3bucket.ts
|
|
1701
|
+
var MAX_DELIVERABLE_EXCERPT_CHARS = 16e4;
|
|
1702
|
+
var MAX_DIFF_EXCERPT_CHARS = 16e4;
|
|
1703
|
+
var MAX_TOTAL_PROMPT_CHARS = 105e4;
|
|
1704
|
+
var FIXED_PROMPT_OVERHEAD_CHARS = 4e4;
|
|
1705
|
+
var PER_VARIANT_PROMPT_OVERHEAD_CHARS = 6e3;
|
|
1706
|
+
var MIN_TOTAL_EXCERPT_CHARS = 5e4;
|
|
1707
|
+
var MAX_EVIDENCE_CANDIDATES = 28;
|
|
1708
|
+
var MAX_FILE_INVENTORY_FILES = 40;
|
|
1709
|
+
var PROMPT_ORDER_V1_EVIDENCE_MULTIPLIER = 1.5;
|
|
1710
|
+
var PROMPT_ORDER_V1_MAX_EVIDENCE_CANDIDATES = 40;
|
|
1711
|
+
var MAX_POETIC_COMPLETION_REASON_CHARS = 1e3;
|
|
1712
|
+
function computePromptOrderEvidenceCandidateLimit(baseLimit, promptOrderMode) {
|
|
1713
|
+
const normalizedBase = Math.max(1, Math.floor(baseLimit || 1));
|
|
1714
|
+
if (promptOrderMode !== "experimental") return normalizedBase;
|
|
1715
|
+
return Math.max(
|
|
1716
|
+
normalizedBase,
|
|
1717
|
+
Math.min(
|
|
1718
|
+
PROMPT_ORDER_V1_MAX_EVIDENCE_CANDIDATES,
|
|
1719
|
+
Math.ceil(normalizedBase * PROMPT_ORDER_V1_EVIDENCE_MULTIPLIER)
|
|
1720
|
+
)
|
|
1721
|
+
);
|
|
1722
|
+
}
|
|
1723
|
+
function applyTotalBudgetCap(caps, variantCount) {
|
|
1724
|
+
const count = Math.max(1, variantCount);
|
|
1725
|
+
const totalOverhead = FIXED_PROMPT_OVERHEAD_CHARS + PER_VARIANT_PROMPT_OVERHEAD_CHARS * count;
|
|
1726
|
+
const excerptBudgetCap = Math.max(
|
|
1727
|
+
MIN_TOTAL_EXCERPT_CHARS,
|
|
1728
|
+
MAX_TOTAL_PROMPT_CHARS - totalOverhead
|
|
1729
|
+
);
|
|
1730
|
+
const totalExcerptBudget = (caps.maxDeliverableExcerptChars + caps.maxDiffExcerptChars) * count;
|
|
1731
|
+
if (totalExcerptBudget <= excerptBudgetCap) return caps;
|
|
1732
|
+
const scale = excerptBudgetCap / totalExcerptBudget;
|
|
1733
|
+
return {
|
|
1734
|
+
maxDeliverableExcerptChars: Math.max(
|
|
1735
|
+
1e3,
|
|
1736
|
+
Math.floor(caps.maxDeliverableExcerptChars * scale)
|
|
1737
|
+
),
|
|
1738
|
+
maxDiffExcerptChars: Math.max(1e3, Math.floor(caps.maxDiffExcerptChars * scale)),
|
|
1739
|
+
maxEvidenceCandidates: caps.maxEvidenceCandidates,
|
|
1740
|
+
maxFileInventoryFiles: caps.maxFileInventoryFiles
|
|
1741
|
+
};
|
|
1742
|
+
}
|
|
1743
|
+
function computeAdaptiveJudgePromptBudgetCaps(variantCount) {
|
|
1744
|
+
const count = Math.max(1, Math.floor(variantCount || 1));
|
|
1745
|
+
let caps;
|
|
1746
|
+
if (count >= 10) {
|
|
1747
|
+
caps = {
|
|
1748
|
+
maxDeliverableExcerptChars: 4e4,
|
|
1749
|
+
maxDiffExcerptChars: 4e4,
|
|
1750
|
+
maxEvidenceCandidates: 12,
|
|
1751
|
+
maxFileInventoryFiles: 12
|
|
1752
|
+
};
|
|
1753
|
+
} else if (count >= 6) {
|
|
1754
|
+
caps = {
|
|
1755
|
+
maxDeliverableExcerptChars: 8e4,
|
|
1756
|
+
maxDiffExcerptChars: 8e4,
|
|
1757
|
+
maxEvidenceCandidates: 16,
|
|
1758
|
+
maxFileInventoryFiles: 16
|
|
1759
|
+
};
|
|
1760
|
+
} else if (count >= 4) {
|
|
1761
|
+
caps = {
|
|
1762
|
+
maxDeliverableExcerptChars: 13e4,
|
|
1763
|
+
maxDiffExcerptChars: 13e4,
|
|
1764
|
+
maxEvidenceCandidates: 20,
|
|
1765
|
+
maxFileInventoryFiles: 20
|
|
1766
|
+
};
|
|
1767
|
+
} else {
|
|
1768
|
+
caps = {
|
|
1769
|
+
maxDeliverableExcerptChars: MAX_DELIVERABLE_EXCERPT_CHARS,
|
|
1770
|
+
maxDiffExcerptChars: MAX_DIFF_EXCERPT_CHARS,
|
|
1771
|
+
maxEvidenceCandidates: MAX_EVIDENCE_CANDIDATES,
|
|
1772
|
+
maxFileInventoryFiles: MAX_FILE_INVENTORY_FILES
|
|
1773
|
+
};
|
|
1774
|
+
}
|
|
1775
|
+
return applyTotalBudgetCap(caps, count);
|
|
1776
|
+
}
|
|
1777
|
+
function buildTruncationNote(fullChars, shownChars, label) {
|
|
1778
|
+
if (fullChars <= shownChars) return "";
|
|
1779
|
+
const pct = (shownChars / fullChars * 100).toFixed(0);
|
|
1780
|
+
return `[${label}: showing ${shownChars.toLocaleString()} of ${fullChars.toLocaleString()} chars (${pct}%)]`;
|
|
1781
|
+
}
|
|
1782
|
+
function normalizeEvidenceTextForMatch(text) {
|
|
1783
|
+
return text.replace(/\r\n/g, "\n").replace(/[“”]/g, '"').replace(/[‘’]/g, "'").replace(/[–—]/g, "-").replace(/\s+/g, " ").trim();
|
|
1784
|
+
}
|
|
1785
|
+
function evidenceQuoteMatchesExcerpt(excerpt, quote) {
|
|
1786
|
+
if (typeof quote !== "string") return false;
|
|
1787
|
+
const trimmed = quote.trim();
|
|
1788
|
+
if (trimmed.length === 0) return false;
|
|
1789
|
+
const canonicalExcerpt = excerpt.replace(/\r\n/g, "\n");
|
|
1790
|
+
if (canonicalExcerpt.includes(trimmed.replace(/\r\n/g, "\n"))) return true;
|
|
1791
|
+
const normalizedExcerpt = normalizeEvidenceTextForMatch(excerpt);
|
|
1792
|
+
const normalizedQuote = normalizeEvidenceTextForMatch(quote);
|
|
1793
|
+
if (normalizedQuote.length === 0) return false;
|
|
1794
|
+
return normalizedExcerpt.includes(normalizedQuote);
|
|
1795
|
+
}
|
|
1796
|
+
function estimatePromptTokens(chars) {
|
|
1797
|
+
return Math.max(0, charsToTokens(chars));
|
|
1798
|
+
}
|
|
1799
|
+
function metricFromText(text) {
|
|
1800
|
+
const chars = text.length;
|
|
1801
|
+
return { chars, estimatedTokens: estimatePromptTokens(chars) };
|
|
1802
|
+
}
|
|
1803
|
+
function metricFromChars(chars) {
|
|
1804
|
+
const safeChars = Math.max(0, Math.floor(chars));
|
|
1805
|
+
return { chars: safeChars, estimatedTokens: estimatePromptTokens(safeChars) };
|
|
1806
|
+
}
|
|
1807
|
+
var DIRECT_IMPL_EVIDENCE_LABEL = "[direct implementation evidence]";
|
|
1808
|
+
var EXECUTOR_CLAIM_LABEL = "[executor claim]";
|
|
1809
|
+
function extractExecutorClaimText(variant) {
|
|
1810
|
+
if (variant.worktreeDiff && variant.worktreeDiff.trim().length > 0) {
|
|
1811
|
+
const { worktreeDiff: _omit, ...withoutDiff } = variant;
|
|
1812
|
+
return sanitizeJudgeText(extractCleanStdout(withoutDiff)).trim();
|
|
1813
|
+
}
|
|
1814
|
+
return sanitizeJudgeText(extractCleanStdout(variant)).trim();
|
|
1815
|
+
}
|
|
1816
|
+
function extractDirectImplementationEvidence(variant) {
|
|
1817
|
+
const diffCandidate = sanitizeJudgeText(extractDiffText(variant)).trim();
|
|
1818
|
+
if (!diffCandidate) return "";
|
|
1819
|
+
if (variant.worktreeDiff?.trim()) return diffCandidate;
|
|
1820
|
+
return /^(?:diff --git |--- (?:a\/|\/dev\/null)|\+\+\+ (?:b\/|\/dev\/null))/m.test(diffCandidate) ? diffCandidate : "";
|
|
1821
|
+
}
|
|
1822
|
+
function sampleEvidenceLines(text, maxCount, candidateClipChars, compactMode, sourceLabel) {
|
|
1823
|
+
if (!text || maxCount <= 0) return [];
|
|
1824
|
+
const candidates = [];
|
|
1825
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1826
|
+
const pushCandidate = (line, lineNo2) => {
|
|
1827
|
+
const trimmed = line.trim();
|
|
1828
|
+
if (!trimmed) return;
|
|
1829
|
+
const clipped = trimmed.length > candidateClipChars ? `${trimmed.slice(0, candidateClipChars - 3)}...` : trimmed;
|
|
1830
|
+
const normalizedKey = normalizeEvidenceTextForMatch(clipped).toLowerCase();
|
|
1831
|
+
if (!normalizedKey) return;
|
|
1832
|
+
if (seen.has(normalizedKey)) return;
|
|
1833
|
+
seen.add(normalizedKey);
|
|
1834
|
+
const body = compactMode ? `L${lineNo2}: ${clipped}` : clipped;
|
|
1835
|
+
candidates.push(sourceLabel ? `${sourceLabel} ${body}` : body);
|
|
1836
|
+
};
|
|
1837
|
+
const allQualifying = [];
|
|
1838
|
+
let lineNo = 0;
|
|
1839
|
+
for (const line of text.split("\n")) {
|
|
1840
|
+
lineNo += 1;
|
|
1841
|
+
const trimmed = line.trim();
|
|
1842
|
+
if (trimmed.length < 8) continue;
|
|
1843
|
+
allQualifying.push({ line: trimmed, lineNo });
|
|
1844
|
+
}
|
|
1845
|
+
if (allQualifying.length === 0) return [];
|
|
1846
|
+
if (allQualifying.length <= maxCount) {
|
|
1847
|
+
for (const entry of allQualifying) {
|
|
1848
|
+
pushCandidate(entry.line, entry.lineNo);
|
|
1849
|
+
}
|
|
1850
|
+
} else {
|
|
1851
|
+
const step = allQualifying.length / maxCount;
|
|
1852
|
+
for (let i = 0; i < maxCount; i++) {
|
|
1853
|
+
const idx = Math.min(Math.floor(i * step), allQualifying.length - 1);
|
|
1854
|
+
const entry = allQualifying[idx];
|
|
1855
|
+
if (!entry) continue;
|
|
1856
|
+
pushCandidate(entry.line, entry.lineNo);
|
|
1857
|
+
if (candidates.length >= maxCount) break;
|
|
1858
|
+
}
|
|
1859
|
+
}
|
|
1860
|
+
return candidates;
|
|
1861
|
+
}
|
|
1862
|
+
function buildVariantEvidenceCandidatesForJudge(variant, options) {
|
|
1863
|
+
const analysisWithoutDiff = variant.executionIntent === "analysis" && !(typeof variant.worktreeDiff === "string" && variant.worktreeDiff.trim().length > 0);
|
|
1864
|
+
const primaryStdout = sanitizeJudgeText(extractCleanStdout(variant)).trim();
|
|
1865
|
+
const claimStdout = extractExecutorClaimText(variant);
|
|
1866
|
+
const cleanedStdout = analysisWithoutDiff ? primaryStdout || claimStdout : claimStdout;
|
|
1867
|
+
const diffText = analysisWithoutDiff ? "" : extractDirectImplementationEvidence(variant);
|
|
1868
|
+
const maxEvidenceCandidates = Math.max(
|
|
1869
|
+
1,
|
|
1870
|
+
Math.floor(options?.maxEvidenceCandidates ?? MAX_EVIDENCE_CANDIDATES)
|
|
1871
|
+
);
|
|
1872
|
+
const compactMode = options?.compactMode === true;
|
|
1873
|
+
const effectiveMaxEvidenceCandidates = compactMode ? Math.max(1, Math.min(4, maxEvidenceCandidates)) : maxEvidenceCandidates;
|
|
1874
|
+
const candidateClipChars = compactMode ? 96 : 220;
|
|
1875
|
+
if (analysisWithoutDiff) {
|
|
1876
|
+
const primary = sampleEvidenceLines(
|
|
1877
|
+
cleanedStdout || diffText,
|
|
1878
|
+
effectiveMaxEvidenceCandidates,
|
|
1879
|
+
candidateClipChars,
|
|
1880
|
+
compactMode,
|
|
1881
|
+
null
|
|
1882
|
+
);
|
|
1883
|
+
return primary.length > 0 ? primary : ["(none)"];
|
|
1884
|
+
}
|
|
1885
|
+
if (diffText.length === 0 && cleanedStdout.length > 0) {
|
|
1886
|
+
const claims = sampleEvidenceLines(
|
|
1887
|
+
cleanedStdout,
|
|
1888
|
+
effectiveMaxEvidenceCandidates,
|
|
1889
|
+
candidateClipChars,
|
|
1890
|
+
compactMode,
|
|
1891
|
+
compactMode ? null : EXECUTOR_CLAIM_LABEL
|
|
1892
|
+
);
|
|
1893
|
+
return claims.length > 0 ? claims : ["(none)"];
|
|
1894
|
+
}
|
|
1895
|
+
if (diffText.length > 0) {
|
|
1896
|
+
const claimBudget = cleanedStdout.length > 0 ? Math.max(1, Math.floor(effectiveMaxEvidenceCandidates / 4)) : 0;
|
|
1897
|
+
const diffBudget = Math.max(1, effectiveMaxEvidenceCandidates - claimBudget);
|
|
1898
|
+
const candidates = sampleEvidenceLines(
|
|
1899
|
+
diffText,
|
|
1900
|
+
diffBudget,
|
|
1901
|
+
candidateClipChars,
|
|
1902
|
+
compactMode,
|
|
1903
|
+
compactMode ? null : DIRECT_IMPL_EVIDENCE_LABEL
|
|
1904
|
+
);
|
|
1905
|
+
if (cleanedStdout.length > 0 && claimBudget > 0) {
|
|
1906
|
+
const claims = sampleEvidenceLines(
|
|
1907
|
+
cleanedStdout,
|
|
1908
|
+
claimBudget,
|
|
1909
|
+
candidateClipChars,
|
|
1910
|
+
compactMode,
|
|
1911
|
+
compactMode ? null : EXECUTOR_CLAIM_LABEL
|
|
1912
|
+
);
|
|
1913
|
+
for (const claim of claims) {
|
|
1914
|
+
if (candidates.length >= effectiveMaxEvidenceCandidates) break;
|
|
1915
|
+
candidates.push(claim);
|
|
1916
|
+
}
|
|
1917
|
+
}
|
|
1918
|
+
return candidates.length > 0 ? candidates : ["(none)"];
|
|
1919
|
+
}
|
|
1920
|
+
if (compactMode && diffText.length > 0) {
|
|
1921
|
+
const primary = sampleEvidenceLines(
|
|
1922
|
+
diffText,
|
|
1923
|
+
effectiveMaxEvidenceCandidates,
|
|
1924
|
+
candidateClipChars,
|
|
1925
|
+
compactMode,
|
|
1926
|
+
null
|
|
1927
|
+
);
|
|
1928
|
+
return primary.length > 0 ? primary : ["(none)"];
|
|
1929
|
+
}
|
|
1930
|
+
return ["(none)"];
|
|
1931
|
+
}
|
|
1932
|
+
function stripWrappedQuoteChars(value) {
|
|
1933
|
+
let trimmed = value.trim();
|
|
1934
|
+
while (trimmed.length >= 2) {
|
|
1935
|
+
const first = trimmed[0];
|
|
1936
|
+
const last = trimmed[trimmed.length - 1];
|
|
1937
|
+
const wraps = first === '"' && last === '"' || first === "'" && last === "'" || first === "`" && last === "`";
|
|
1938
|
+
if (!wraps) break;
|
|
1939
|
+
trimmed = trimmed.slice(1, -1).trim();
|
|
1940
|
+
}
|
|
1941
|
+
return trimmed;
|
|
1942
|
+
}
|
|
1943
|
+
function normalizePathForMatch(pathText) {
|
|
1944
|
+
return pathText.replace(/\\/g, "/").replace(/^\.\//, "");
|
|
1945
|
+
}
|
|
1946
|
+
function extractEvidencePathReference(rawQuote) {
|
|
1947
|
+
const normalized = stripWrappedQuoteChars(rawQuote).replace(/^file:\s*/i, "").trim();
|
|
1948
|
+
if (!normalized) return null;
|
|
1949
|
+
const fileRefMatch = normalized.match(
|
|
1950
|
+
/^((?:[a-zA-Z]:[\\/])?[a-zA-Z0-9_.-]+(?:[\\/][a-zA-Z0-9_.-]+)+\.[a-zA-Z0-9]{1,8})(?::\d+(?::\d+)?)?(?:#L\d+(?:C\d+)?)?$/i
|
|
1951
|
+
);
|
|
1952
|
+
if (!fileRefMatch?.[1]) return null;
|
|
1953
|
+
return normalizePathForMatch(fileRefMatch[1]);
|
|
1954
|
+
}
|
|
1955
|
+
function excerptContainsPathReference(excerpt, pathReference) {
|
|
1956
|
+
const normalizedExcerpt = excerpt.replace(/\\/g, "/");
|
|
1957
|
+
const candidates = /* @__PURE__ */ new Set([pathReference]);
|
|
1958
|
+
const withoutDiffPrefix = pathReference.startsWith("a/") || pathReference.startsWith("b/") ? pathReference.slice(2) : pathReference;
|
|
1959
|
+
candidates.add(withoutDiffPrefix);
|
|
1960
|
+
candidates.add(`a/${withoutDiffPrefix}`);
|
|
1961
|
+
candidates.add(`b/${withoutDiffPrefix}`);
|
|
1962
|
+
for (const candidate of candidates) {
|
|
1963
|
+
if (candidate && normalizedExcerpt.includes(candidate)) {
|
|
1964
|
+
return true;
|
|
1965
|
+
}
|
|
1966
|
+
}
|
|
1967
|
+
return false;
|
|
1968
|
+
}
|
|
1969
|
+
function evidenceQuoteIsValid(excerpt, candidates, quote) {
|
|
1970
|
+
const raw = typeof quote === "string" ? quote.trim() : "";
|
|
1971
|
+
if (!raw) return false;
|
|
1972
|
+
const idMatch = raw.match(/^E(\d{1,2})$/i);
|
|
1973
|
+
if (idMatch) {
|
|
1974
|
+
const idx = Number(idMatch[1]);
|
|
1975
|
+
if (!Number.isFinite(idx) || idx < 1 || idx > candidates.length) return false;
|
|
1976
|
+
return true;
|
|
1977
|
+
}
|
|
1978
|
+
const pathReference = extractEvidencePathReference(raw);
|
|
1979
|
+
if (pathReference) {
|
|
1980
|
+
if (excerptContainsPathReference(excerpt, pathReference)) {
|
|
1981
|
+
return true;
|
|
1982
|
+
}
|
|
1983
|
+
const candidateHasPath = candidates.some(
|
|
1984
|
+
(candidate) => normalizeEvidenceTextForMatch(candidate).toLowerCase().includes(pathReference.toLowerCase())
|
|
1985
|
+
);
|
|
1986
|
+
if (candidateHasPath) {
|
|
1987
|
+
return true;
|
|
1988
|
+
}
|
|
1989
|
+
}
|
|
1990
|
+
const normalizedQuote = normalizeEvidenceTextForMatch(raw).toLowerCase();
|
|
1991
|
+
if (!normalizedQuote) return false;
|
|
1992
|
+
for (const candidate of candidates) {
|
|
1993
|
+
if (normalizeEvidenceTextForMatch(candidate).toLowerCase() === normalizedQuote) return true;
|
|
1994
|
+
}
|
|
1995
|
+
if (normalizedQuote.replace(/\s+/g, "").length < 10) {
|
|
1996
|
+
return false;
|
|
1997
|
+
}
|
|
1998
|
+
return evidenceQuoteMatchesExcerpt(excerpt, raw);
|
|
1999
|
+
}
|
|
2000
|
+
function sanitizeJudgeText(text) {
|
|
2001
|
+
if (!text) return "";
|
|
2002
|
+
let out = text.replace(/\uE200cite\uE202[^\uE201]*\uE201/g, "");
|
|
2003
|
+
out = presentUntrustedJudgeEvidence(out);
|
|
2004
|
+
out = out.split("\n").flatMap((line) => {
|
|
2005
|
+
if (!line.includes('"type":"poetic_completion"')) return [line];
|
|
2006
|
+
try {
|
|
2007
|
+
const parsed = JSON.parse(line.trim());
|
|
2008
|
+
if (parsed.type === "poetic_completion" && typeof parsed.reason === "string") {
|
|
2009
|
+
const editIntent = typeof parsed.editIntent === "string" ? parsed.editIntent : "unknown";
|
|
2010
|
+
const filesEdited = Array.isArray(parsed.filesEdited) ? parsed.filesEdited.length : 0;
|
|
2011
|
+
const reason = sanitizePoeticCompletionReason(parsed.reason);
|
|
2012
|
+
return [
|
|
2013
|
+
`Poetic completion signal: editIntent=${editIntent}; filesEdited=${filesEdited}; reason=${reason}`
|
|
2014
|
+
];
|
|
2015
|
+
}
|
|
2016
|
+
} catch {
|
|
2017
|
+
}
|
|
2018
|
+
return [];
|
|
2019
|
+
}).join("\n");
|
|
2020
|
+
return out;
|
|
2021
|
+
}
|
|
2022
|
+
function sanitizePoeticCompletionReason(reason) {
|
|
2023
|
+
const withoutControlChars = Array.from(reason).filter((char) => {
|
|
2024
|
+
const code = char.charCodeAt(0);
|
|
2025
|
+
return code === 9 || code === 10 || code === 13 || code >= 32 && code !== 127;
|
|
2026
|
+
}).join("");
|
|
2027
|
+
return truncateHeadTail(
|
|
2028
|
+
redactSuspiciousJudgePromptPatterns(withoutControlChars),
|
|
2029
|
+
MAX_POETIC_COMPLETION_REASON_CHARS
|
|
2030
|
+
);
|
|
2031
|
+
}
|
|
2032
|
+
function truncateHeadTail(text, maxChars) {
|
|
2033
|
+
if (text.length <= maxChars) return text;
|
|
2034
|
+
const marker = "\n...[truncated]...\n";
|
|
2035
|
+
const remaining = maxChars - marker.length;
|
|
2036
|
+
const head = Math.max(0, Math.floor(remaining * 0.6));
|
|
2037
|
+
const tail = Math.max(0, remaining - head);
|
|
2038
|
+
return `${text.slice(0, head)}${marker}${text.slice(Math.max(0, text.length - tail))}`;
|
|
2039
|
+
}
|
|
2040
|
+
function splitDiffByFile(text) {
|
|
2041
|
+
const sections = [];
|
|
2042
|
+
const diffPattern = /^diff --git a\/(.+?) b\//gm;
|
|
2043
|
+
const starts = [];
|
|
2044
|
+
for (; ; ) {
|
|
2045
|
+
const match = diffPattern.exec(text);
|
|
2046
|
+
if (!match) break;
|
|
2047
|
+
const path3 = match[1];
|
|
2048
|
+
if (!path3) continue;
|
|
2049
|
+
starts.push({ path: path3, index: match.index });
|
|
2050
|
+
}
|
|
2051
|
+
if (starts.length === 0) {
|
|
2052
|
+
const changedLines = text.split("\n").filter(
|
|
2053
|
+
(line) => line.startsWith("+") && !line.startsWith("+++") || line.startsWith("-") && !line.startsWith("---")
|
|
2054
|
+
).length;
|
|
2055
|
+
return [
|
|
2056
|
+
{
|
|
2057
|
+
path: "(single)",
|
|
2058
|
+
text,
|
|
2059
|
+
lineCount: text.split("\n").length,
|
|
2060
|
+
changedLines,
|
|
2061
|
+
originalIndex: 0
|
|
2062
|
+
}
|
|
2063
|
+
];
|
|
2064
|
+
}
|
|
2065
|
+
for (let i = 0; i < starts.length; i++) {
|
|
2066
|
+
const current = starts[i];
|
|
2067
|
+
if (!current) continue;
|
|
2068
|
+
const start = current.index;
|
|
2069
|
+
const next = starts[i + 1];
|
|
2070
|
+
const end = next ? next.index : text.length;
|
|
2071
|
+
const sectionText = text.slice(start, end);
|
|
2072
|
+
const changedLines = sectionText.split("\n").filter(
|
|
2073
|
+
(line) => line.startsWith("+") && !line.startsWith("+++") || line.startsWith("-") && !line.startsWith("---")
|
|
2074
|
+
).length;
|
|
2075
|
+
sections.push({
|
|
2076
|
+
path: current.path,
|
|
2077
|
+
text: sectionText,
|
|
2078
|
+
lineCount: sectionText.split("\n").length,
|
|
2079
|
+
changedLines,
|
|
2080
|
+
originalIndex: i
|
|
2081
|
+
});
|
|
2082
|
+
}
|
|
2083
|
+
return sections;
|
|
2084
|
+
}
|
|
2085
|
+
function truncateWithFileBoundaries(text, maxChars, options) {
|
|
2086
|
+
if (text.length <= maxChars) return text;
|
|
2087
|
+
const fileSections = splitDiffByFile(text);
|
|
2088
|
+
if (fileSections.length <= 1) {
|
|
2089
|
+
return truncateHeadTail(text, maxChars);
|
|
2090
|
+
}
|
|
2091
|
+
const priorityPaths = (options?.priorityPaths ?? []).map(
|
|
2092
|
+
(path3) => normalizePathForMatch(path3).replace(/^(a|b)\//, "")
|
|
2093
|
+
);
|
|
2094
|
+
const rankedSections = [...fileSections].sort((left, right) => {
|
|
2095
|
+
const leftPath = normalizePathForMatch(left.path).replace(/^(a|b)\//, "");
|
|
2096
|
+
const rightPath = normalizePathForMatch(right.path).replace(/^(a|b)\//, "");
|
|
2097
|
+
const leftPriority = priorityPaths.some(
|
|
2098
|
+
(priority) => leftPath === priority || leftPath.endsWith(`/${priority}`)
|
|
2099
|
+
) ? 1 : 0;
|
|
2100
|
+
const rightPriority = priorityPaths.some(
|
|
2101
|
+
(priority) => rightPath === priority || rightPath.endsWith(`/${priority}`)
|
|
2102
|
+
) ? 1 : 0;
|
|
2103
|
+
if (leftPriority !== rightPriority) return rightPriority - leftPriority;
|
|
2104
|
+
const leftDensity = left.changedLines / Math.max(1, left.lineCount);
|
|
2105
|
+
const rightDensity = right.changedLines / Math.max(1, right.lineCount);
|
|
2106
|
+
if (leftDensity !== rightDensity) return rightDensity - leftDensity;
|
|
2107
|
+
return left.originalIndex - right.originalIndex;
|
|
2108
|
+
});
|
|
2109
|
+
const included = [];
|
|
2110
|
+
const omitted = [];
|
|
2111
|
+
const summaryBudget = 200;
|
|
2112
|
+
let remaining = maxChars - summaryBudget;
|
|
2113
|
+
for (const section of rankedSections) {
|
|
2114
|
+
if (section.text.length <= remaining) {
|
|
2115
|
+
included.push(section.text);
|
|
2116
|
+
remaining -= section.text.length;
|
|
2117
|
+
} else if (included.length === 0) {
|
|
2118
|
+
included.push(truncateHeadTail(section.text, remaining));
|
|
2119
|
+
remaining = 0;
|
|
2120
|
+
} else {
|
|
2121
|
+
omitted.push({ path: section.path, lines: section.lineCount });
|
|
2122
|
+
}
|
|
2123
|
+
}
|
|
2124
|
+
if (omitted.length > 0) {
|
|
2125
|
+
const omittedSummary = `
|
|
2126
|
+
[${omitted.length} file(s) omitted: ${omitted.map((f) => `${f.path} (+${f.lines})`).join(", ")}]
|
|
2127
|
+
`;
|
|
2128
|
+
included.push(omittedSummary);
|
|
2129
|
+
if (priorityPaths.length > 0) {
|
|
2130
|
+
const omittedPriority = omitted.filter((f) => {
|
|
2131
|
+
const normalizedPath = normalizePathForMatch(f.path).replace(/^(a|b)\//, "");
|
|
2132
|
+
return priorityPaths.some((p) => normalizedPath === p || normalizedPath.endsWith(`/${p}`));
|
|
2133
|
+
});
|
|
2134
|
+
if (omittedPriority.length > 0) {
|
|
2135
|
+
included.push(
|
|
2136
|
+
`
|
|
2137
|
+
[WARNING: ${omittedPriority.length} acceptance-criteria file(s) omitted due to size: ${omittedPriority.map((f) => f.path).join(", ")}. Evidence for these files is incomplete.]
|
|
2138
|
+
`
|
|
2139
|
+
);
|
|
2140
|
+
}
|
|
2141
|
+
}
|
|
2142
|
+
}
|
|
2143
|
+
return included.join("");
|
|
2144
|
+
}
|
|
2145
|
+
function buildVariantDeliverableExcerptForJudge(variant, options) {
|
|
2146
|
+
const analysisWithoutDiff = variant.executionIntent === "analysis" && !(typeof variant.worktreeDiff === "string" && variant.worktreeDiff.trim().length > 0);
|
|
2147
|
+
const primaryStdout = sanitizeJudgeText(extractCleanStdout(variant)).trim();
|
|
2148
|
+
const claimStdout = extractExecutorClaimText(variant);
|
|
2149
|
+
const cleanedStdout = analysisWithoutDiff ? primaryStdout || claimStdout : claimStdout;
|
|
2150
|
+
const diffText = analysisWithoutDiff ? "" : extractDirectImplementationEvidence(variant);
|
|
2151
|
+
const maxDeliverableExcerptChars = Math.max(
|
|
2152
|
+
1,
|
|
2153
|
+
Math.floor(options?.maxDeliverableExcerptChars ?? MAX_DELIVERABLE_EXCERPT_CHARS)
|
|
2154
|
+
);
|
|
2155
|
+
const maxDiffExcerptChars = Math.max(
|
|
2156
|
+
1,
|
|
2157
|
+
Math.floor(options?.maxDiffExcerptChars ?? MAX_DIFF_EXCERPT_CHARS)
|
|
2158
|
+
);
|
|
2159
|
+
const truncation = {
|
|
2160
|
+
outputFullChars: cleanedStdout.length,
|
|
2161
|
+
outputShownChars: Math.min(cleanedStdout.length, maxDeliverableExcerptChars),
|
|
2162
|
+
diffFullChars: diffText.length,
|
|
2163
|
+
diffShownChars: Math.min(diffText.length, maxDiffExcerptChars)
|
|
2164
|
+
};
|
|
2165
|
+
const structuralRisks = scanDiffForStructuralRisks(diffText);
|
|
2166
|
+
const priorityFiles = Array.from(
|
|
2167
|
+
/* @__PURE__ */ new Set([
|
|
2168
|
+
...options?.priorityFiles ?? [],
|
|
2169
|
+
...buildVariantPriorityFiles(variant, structuralRisks)
|
|
2170
|
+
])
|
|
2171
|
+
);
|
|
2172
|
+
const timedOutWithPartialEdits = variant.timedOut === true && variant.partialEdits === true;
|
|
2173
|
+
const parts = [];
|
|
2174
|
+
if (diffText.length > 0) {
|
|
2175
|
+
const diffLabel = timedOutWithPartialEdits ? "<<<DELIVERABLE EXCERPT: DIFF [PARTIAL/INCOMPLETE]>>>" : "<<<DELIVERABLE EXCERPT: DIFF>>>";
|
|
2176
|
+
parts.push(
|
|
2177
|
+
`${diffLabel}
|
|
2178
|
+
Source: direct implementation evidence
|
|
2179
|
+
${frameUntrustedJudgeEvidence(truncateWithFileBoundaries(diffText, maxDiffExcerptChars, { priorityPaths: priorityFiles }))}
|
|
2180
|
+
<<<END EXCERPT>>>`
|
|
2181
|
+
);
|
|
2182
|
+
}
|
|
2183
|
+
if (cleanedStdout.length > 0) {
|
|
2184
|
+
if (!analysisWithoutDiff) {
|
|
2185
|
+
parts.push(
|
|
2186
|
+
`<<<UNTRUSTED DELIVERABLE EXCERPT: PROVIDER OUTPUT>>>
|
|
2187
|
+
Source: executor claim (not implementation evidence)
|
|
2188
|
+
${frameUntrustedJudgeEvidence(truncateHeadTail(cleanedStdout, maxDeliverableExcerptChars))}
|
|
2189
|
+
<<<END UNTRUSTED EXCERPT>>>`
|
|
2190
|
+
);
|
|
2191
|
+
} else {
|
|
2192
|
+
parts.push(
|
|
2193
|
+
`<<<UNTRUSTED DELIVERABLE EXCERPT: PROVIDER OUTPUT>>>
|
|
2194
|
+
${frameUntrustedJudgeEvidence(truncateHeadTail(cleanedStdout, maxDeliverableExcerptChars))}
|
|
2195
|
+
<<<END UNTRUSTED EXCERPT>>>`
|
|
2196
|
+
);
|
|
2197
|
+
}
|
|
2198
|
+
}
|
|
2199
|
+
const text = parts.length > 0 ? parts.join("\n\n") : "<<<DELIVERABLE EXCERPT: NONE>>>\n(none)\n<<<END EXCERPT>>>";
|
|
2200
|
+
return { text, truncation };
|
|
2201
|
+
}
|
|
2202
|
+
function buildVariantPriorityFiles(variant, structuralRisks) {
|
|
2203
|
+
const priorities = /* @__PURE__ */ new Set();
|
|
2204
|
+
for (const file of variant.testsExecuted?.filesTested ?? []) {
|
|
2205
|
+
priorities.add(normalizePathForMatch(file).replace(/^(a|b)\//, ""));
|
|
2206
|
+
}
|
|
2207
|
+
for (const finding of structuralRisks.findings) {
|
|
2208
|
+
if (finding.filePath) {
|
|
2209
|
+
priorities.add(normalizePathForMatch(finding.filePath).replace(/^(a|b)\//, ""));
|
|
2210
|
+
}
|
|
2211
|
+
}
|
|
2212
|
+
return Array.from(priorities);
|
|
2213
|
+
}
|
|
2214
|
+
function formatEmpiricalSignals(signals, variant) {
|
|
2215
|
+
const lines = [];
|
|
2216
|
+
lines.push("**Execution**:");
|
|
2217
|
+
lines.push(` success: ${signals.execution.success}`);
|
|
2218
|
+
lines.push(` emptyChanges: ${signals.execution.emptyChanges}`);
|
|
2219
|
+
lines.push(` timedOut: ${signals.execution.timedOut}`);
|
|
2220
|
+
lines.push(` partialEdits: ${signals.execution.partialEdits === true}`);
|
|
2221
|
+
if (signals.execution.failureType) {
|
|
2222
|
+
lines.push(` failureType: ${signals.execution.failureType}`);
|
|
2223
|
+
}
|
|
2224
|
+
if (signals.execution.failureReason) {
|
|
2225
|
+
lines.push(` failureReason: ${signals.execution.failureReason}`);
|
|
2226
|
+
}
|
|
2227
|
+
if (signals.execution.timedOut && signals.execution.partialEdits) {
|
|
2228
|
+
lines.push(
|
|
2229
|
+
" note: timed out after partial implementation; score partial credit from diff evidence"
|
|
2230
|
+
);
|
|
2231
|
+
}
|
|
2232
|
+
lines.push(` taskKind: ${signals.execution.taskKind}`);
|
|
2233
|
+
if (signals.execution.gitTelemetryValidation?.valid === false) {
|
|
2234
|
+
lines.push(" gitTelemetryValidation: failed");
|
|
2235
|
+
for (const error of signals.execution.gitTelemetryValidation.errors.slice(0, 3)) {
|
|
2236
|
+
lines.push(` error: ${error}`);
|
|
2237
|
+
}
|
|
2238
|
+
}
|
|
2239
|
+
if (signals.execution.targetIntegrityEvidence?.fabricatedTargets.length) {
|
|
2240
|
+
const evidence = signals.execution.targetIntegrityEvidence;
|
|
2241
|
+
lines.push(" targetIntegrityEvidence: fabricated FILE target(s) detected");
|
|
2242
|
+
lines.push(` fabricatedTargets: [${evidence.fabricatedTargets.join(", ")}]`);
|
|
2243
|
+
lines.push(` rawCompleteness: ${evidence.rawCompleteness}`);
|
|
2244
|
+
lines.push(` suggestedCompleteness: ${evidence.suggestedCompleteness}`);
|
|
2245
|
+
lines.push(` reason: ${evidence.reason}`);
|
|
2246
|
+
}
|
|
2247
|
+
if (signals.execution.testStatus) {
|
|
2248
|
+
lines.push(` testStatus (raw, pre-baseline): ${signals.execution.testStatus}`);
|
|
2249
|
+
}
|
|
2250
|
+
if (signals.execution.verificationStatusAdjusted) {
|
|
2251
|
+
lines.push(
|
|
2252
|
+
` verificationStatus (adjusted, baseline-aware): ${signals.execution.verificationStatusAdjusted}`
|
|
2253
|
+
);
|
|
2254
|
+
}
|
|
2255
|
+
const verificationIntent = signals.verificationIntent;
|
|
2256
|
+
lines.push("\n**Verification Intent**:");
|
|
2257
|
+
lines.push(` expectation: ${verificationIntent?.type ?? "neither"}`);
|
|
2258
|
+
lines.push(` confidence: ${(verificationIntent?.confidence ?? 0).toFixed(2)}`);
|
|
2259
|
+
if (verificationIntent?.evidence?.length) {
|
|
2260
|
+
lines.push(` evidence: [${verificationIntent.evidence.join(" | ")}]`);
|
|
2261
|
+
}
|
|
2262
|
+
if (verificationIntent?.overrideSource) {
|
|
2263
|
+
lines.push(` override: ${verificationIntent.overrideSource}`);
|
|
2264
|
+
}
|
|
2265
|
+
lines.push("\n**File Targets**:");
|
|
2266
|
+
lines.push(` fileTargets: [${signals.fileTargets.join(", ")}]`);
|
|
2267
|
+
lines.push(` changedFiles: [${signals.changedFiles.join(", ")}]`);
|
|
2268
|
+
lines.push(` targetCompliance: ${signals.targetCompliance}`);
|
|
2269
|
+
if (signals.offTargetFiles.length > 0) {
|
|
2270
|
+
lines.push(` offTargetFiles: [${signals.offTargetFiles.join(", ")}]`);
|
|
2271
|
+
}
|
|
2272
|
+
if (signals.deliverableTargets && signals.deliverableTargets.length > 0) {
|
|
2273
|
+
lines.push("\n**Deliverable Requirements**:");
|
|
2274
|
+
lines.push(` deliverableTargets: [${signals.deliverableTargets.join(", ")}]`);
|
|
2275
|
+
lines.push(` deliverableSatisfied: ${signals.deliverableSatisfied}`);
|
|
2276
|
+
if (signals.deliverablePossibleCollision) {
|
|
2277
|
+
lines.push(
|
|
2278
|
+
' warning: DELIVERABLE marker may have been a prose "Deliverables:" section (non-uppercase marker)'
|
|
2279
|
+
);
|
|
2280
|
+
}
|
|
2281
|
+
}
|
|
2282
|
+
const changedTestFiles = signals.changedFiles.filter((p) => isProjectTestPath(p));
|
|
2283
|
+
if (changedTestFiles.length > 0) {
|
|
2284
|
+
lines.push(` changedTestFiles: [${changedTestFiles.join(", ")}]`);
|
|
2285
|
+
}
|
|
2286
|
+
lines.push("\n**Diff Stats**:");
|
|
2287
|
+
lines.push(` filesChanged: ${signals.diffStats.filesChanged}`);
|
|
2288
|
+
lines.push(` linesAdded: ${signals.diffStats.linesAdded}`);
|
|
2289
|
+
lines.push(` linesRemoved: ${signals.diffStats.linesRemoved}`);
|
|
2290
|
+
if (signals.execution.workspaceEditsRequested > 0 || signals.execution.workspaceEditsApplied > 0) {
|
|
2291
|
+
lines.push("\n**Workspace Edits**:");
|
|
2292
|
+
lines.push(` requested: ${signals.execution.workspaceEditsRequested}`);
|
|
2293
|
+
lines.push(` applied: ${signals.execution.workspaceEditsApplied}`);
|
|
2294
|
+
if (signals.execution.ws007Count > 0) {
|
|
2295
|
+
lines.push(` ws007Rejected: ${signals.execution.ws007Count}`);
|
|
2296
|
+
}
|
|
2297
|
+
if (signals.execution.workspaceEditsApplied === 0 && (signals.diffStats.filesChanged > 0 || signals.diffStats.linesAdded > 0 || signals.diffStats.linesRemoved > 0)) {
|
|
2298
|
+
lines.push(
|
|
2299
|
+
" note: JSONL applied=0 but git diff shows changes; treat as applied for evaluation"
|
|
2300
|
+
);
|
|
2301
|
+
} else if (signals.execution.workspaceEditsRequested > signals.execution.workspaceEditsApplied) {
|
|
2302
|
+
lines.push(" note: partial workspace edits applied");
|
|
2303
|
+
}
|
|
2304
|
+
if (signals.execution.ws007Count > 10) {
|
|
2305
|
+
lines.push(
|
|
2306
|
+
` HIGH WS-007 REJECTIONS (${signals.execution.ws007Count}): This variant's output may be incomplete. Many file writes were blocked by the empty-contents safety guard. Score the variant on what it delivered, but note potential handicap.`
|
|
2307
|
+
);
|
|
2308
|
+
}
|
|
2309
|
+
}
|
|
2310
|
+
if (signals.testMetrics) {
|
|
2311
|
+
lines.push("\n**Test Metrics**:");
|
|
2312
|
+
lines.push(` totalTests: ${signals.testMetrics.totalTests}`);
|
|
2313
|
+
lines.push(` testQualityLevel: ${signals.testMetrics.testQualityLevel}`);
|
|
2314
|
+
lines.push(` testQualityScore: ${signals.testMetrics.testQualityScore}/15`);
|
|
2315
|
+
lines.push(
|
|
2316
|
+
` commandPathTests: ${signals.testMetrics.commandPathTests.count} (confidence: ${signals.testMetrics.commandPathTests.confidence})`
|
|
2317
|
+
);
|
|
2318
|
+
lines.push(` exitCodeAssertions: ${signals.testMetrics.exitCodeAssertions.count}`);
|
|
2319
|
+
lines.push(` stderrAssertions: ${signals.testMetrics.stderrAssertions.count}`);
|
|
2320
|
+
}
|
|
2321
|
+
if (signals.requirements?.found) {
|
|
2322
|
+
lines.push("\n**Requirements**:");
|
|
2323
|
+
if (signals.requirements.parseErrors.length > 0) {
|
|
2324
|
+
lines.push(` parseErrors: [${signals.requirements.parseErrors.join(" | ")}]`);
|
|
2325
|
+
}
|
|
2326
|
+
lines.push(` compliant: ${signals.requirements.compliant}`);
|
|
2327
|
+
lines.push(
|
|
2328
|
+
` satisfied: ${signals.requirements.satisfiedCount}/${signals.requirements.totalCount}`
|
|
2329
|
+
);
|
|
2330
|
+
lines.push(` penalty: ${signals.requirements.penalty}`);
|
|
2331
|
+
if (signals.requirements.violations.length > 0) {
|
|
2332
|
+
lines.push(` violations: [${signals.requirements.violations.join(" | ")}]`);
|
|
2333
|
+
}
|
|
2334
|
+
const total = signals.requirements.totalCount;
|
|
2335
|
+
const satisfied = signals.requirements.satisfiedCount;
|
|
2336
|
+
if (total > 0 && satisfied / total < 0.5) {
|
|
2337
|
+
lines.push(
|
|
2338
|
+
` LOW COVERAGE: Only ${satisfied}/${total} requirements satisfied. Completeness score should reflect this gap.`
|
|
2339
|
+
);
|
|
2340
|
+
}
|
|
2341
|
+
}
|
|
2342
|
+
if (variant) {
|
|
2343
|
+
const structuralRisks = scanDiffForStructuralRisks(sanitizeJudgeText(extractDiffText(variant)));
|
|
2344
|
+
if (structuralRisks.findings.length > 0) {
|
|
2345
|
+
lines.push("\n**Structural Risk Signals**:");
|
|
2346
|
+
lines.push(
|
|
2347
|
+
` critical: ${structuralRisks.criticalCount}, major: ${structuralRisks.majorCount}, minor: ${structuralRisks.minorCount}`
|
|
2348
|
+
);
|
|
2349
|
+
for (const finding of structuralRisks.findings.slice(0, 3)) {
|
|
2350
|
+
const displayPath = finding.filePath !== void 0 ? escapePathForJudgePromptDisplay(finding.filePath) : void 0;
|
|
2351
|
+
const location = displayPath && finding.line ? `${displayPath}:${finding.line}` : displayPath ?? "unknown location";
|
|
2352
|
+
lines.push(` - ${finding.severity.toUpperCase()}: ${finding.summary} (${location})`);
|
|
2353
|
+
}
|
|
2354
|
+
}
|
|
2355
|
+
const testFailureDetails = extractFailureDetailSnippet(variant, "test");
|
|
2356
|
+
if (testFailureDetails) {
|
|
2357
|
+
lines.push("\n**Test Failure Details**:");
|
|
2358
|
+
lines.push(...testFailureDetails.split("\n").map((line) => ` ${line}`));
|
|
2359
|
+
}
|
|
2360
|
+
const buildFailureDetails = extractFailureDetailSnippet(variant, "build");
|
|
2361
|
+
if (buildFailureDetails) {
|
|
2362
|
+
lines.push("\n**Build Failure Details**:");
|
|
2363
|
+
lines.push(...buildFailureDetails.split("\n").map((line) => ` ${line}`));
|
|
2364
|
+
}
|
|
2365
|
+
}
|
|
2366
|
+
if (signals.testGate) {
|
|
2367
|
+
lines.push("\n**Test Gate (Verified)**:");
|
|
2368
|
+
lines.push(` testStatus (raw): ${signals.testGate.testStatus}`);
|
|
2369
|
+
lines.push(` testsRun: ${signals.testGate.testsRun}`);
|
|
2370
|
+
lines.push(` testsPassed: ${signals.testGate.testsPassed}`);
|
|
2371
|
+
lines.push(` testsFailed: ${signals.testGate.testsFailed}`);
|
|
2372
|
+
if (typeof signals.testGate.exitCode === "number") {
|
|
2373
|
+
lines.push(` exitCode: ${signals.testGate.exitCode}`);
|
|
2374
|
+
}
|
|
2375
|
+
if (signals.testGate.signalConsistency) {
|
|
2376
|
+
lines.push(` signalConsistency: ${signals.testGate.signalConsistency}`);
|
|
2377
|
+
}
|
|
2378
|
+
}
|
|
2379
|
+
if (variant && hasVerificationEvidenceForQualityCard(variant, signals)) {
|
|
2380
|
+
lines.push("");
|
|
2381
|
+
lines.push(formatVerificationEvidenceQualitySummary(variant));
|
|
2382
|
+
}
|
|
2383
|
+
if (JUDGE_FEATURE_FLAGS.VARIANT_DIFFERENTIATION_SIGNALS === "on" && signals.scopeViolation && signals.scopeViolation.scopeViolationSeverity !== "none") {
|
|
2384
|
+
const sv = signals.scopeViolation;
|
|
2385
|
+
lines.push("\n**Scope Violation**:");
|
|
2386
|
+
lines.push(` severity: ${sv.scopeViolationSeverity}`);
|
|
2387
|
+
lines.push(` scopeSource: ${sv.scopeSource}`);
|
|
2388
|
+
lines.push(` outOfScopeRatio: ${sv.outOfScopeRatio.toFixed(2)}`);
|
|
2389
|
+
if (sv.outOfScopeFiles.length > 0) {
|
|
2390
|
+
lines.push(` outOfScopeFiles: [${sv.outOfScopeFiles.slice(0, 5).join(", ")}]`);
|
|
2391
|
+
}
|
|
2392
|
+
if (sv.highRiskOutOfScope.length > 0) {
|
|
2393
|
+
lines.push(` highRiskOutOfScope: [${sv.highRiskOutOfScope.join(", ")}]`);
|
|
2394
|
+
}
|
|
2395
|
+
}
|
|
2396
|
+
if (JUDGE_FEATURE_FLAGS.VARIANT_DIFFERENTIATION_SIGNALS === "on" && signals.coherenceCheck && !signals.coherenceCheck.coherent) {
|
|
2397
|
+
lines.push("\n**Coherence Check**:");
|
|
2398
|
+
lines.push(` coherent: ${signals.coherenceCheck.coherent}`);
|
|
2399
|
+
for (const issue of signals.coherenceCheck.issues) {
|
|
2400
|
+
lines.push(` - ${issue.check}: ${issue.detail}`);
|
|
2401
|
+
}
|
|
2402
|
+
}
|
|
2403
|
+
return lines.join("\n");
|
|
2404
|
+
}
|
|
2405
|
+
function extractFailureDetailSnippet(variant, kind, maxChars = 500) {
|
|
2406
|
+
const findings = variant.verification?.findings ?? [];
|
|
2407
|
+
const matching = findings.filter((finding) => finding.kind === kind && typeof finding.outputTail === "string").map((finding) => finding.outputTail?.trim() ?? "").filter((output) => output.length > 0);
|
|
2408
|
+
if (matching.length === 0) return null;
|
|
2409
|
+
return clipSnippet(matching.join("\n"), maxChars);
|
|
2410
|
+
}
|
|
2411
|
+
function hasVerificationEvidenceForQualityCard(variant, signals) {
|
|
2412
|
+
return Boolean(
|
|
2413
|
+
variant.verification || variant.verificationLoop?.enabled || variant.testsExecuted || signals.buildGate || signals.testGate
|
|
2414
|
+
);
|
|
2415
|
+
}
|
|
2416
|
+
function clipSnippet(text, maxChars = 500) {
|
|
2417
|
+
if (typeof text !== "string") return null;
|
|
2418
|
+
const trimmed = text.trim();
|
|
2419
|
+
if (!trimmed) return null;
|
|
2420
|
+
if (trimmed.length <= maxChars) return trimmed;
|
|
2421
|
+
return `${trimmed.slice(0, maxChars)}...`;
|
|
2422
|
+
}
|
|
2423
|
+
function buildFileInventory(variant, options) {
|
|
2424
|
+
const signals = variant.empiricalSignals;
|
|
2425
|
+
if (!signals || signals.changedFiles.length === 0) return "";
|
|
2426
|
+
const files = signals.changedFiles;
|
|
2427
|
+
const maxFileInventoryFiles = Math.max(
|
|
2428
|
+
1,
|
|
2429
|
+
Math.floor(options?.maxFileInventoryFiles ?? MAX_FILE_INVENTORY_FILES)
|
|
2430
|
+
);
|
|
2431
|
+
const lines = [`**Files Changed (${files.length}):**`];
|
|
2432
|
+
for (const f of files.slice(0, maxFileInventoryFiles)) {
|
|
2433
|
+
lines.push(` - ${f}`);
|
|
2434
|
+
}
|
|
2435
|
+
if (files.length > maxFileInventoryFiles) {
|
|
2436
|
+
lines.push(` - ... and ${files.length - maxFileInventoryFiles} more`);
|
|
2437
|
+
}
|
|
2438
|
+
return lines.join("\n");
|
|
2439
|
+
}
|
|
2440
|
+
function resolveAcceptanceCriteriaTargets(taskPrompt) {
|
|
2441
|
+
const spec = parsePromptSpec(taskPrompt);
|
|
2442
|
+
const inlineTargets = Array.from(taskPrompt.matchAll(/\bFILE:\s*([^\s]+)/gi)).map(
|
|
2443
|
+
(m) => m[1] ?? ""
|
|
2444
|
+
);
|
|
2445
|
+
return Array.from(
|
|
2446
|
+
new Set([...spec.fileConstraint?.targets ?? [], ...inlineTargets].filter(Boolean))
|
|
2447
|
+
);
|
|
2448
|
+
}
|
|
2449
|
+
function buildAIPrompt3Bucket(variants, taskPrompt, taskKind, options) {
|
|
2450
|
+
const variantSections = [];
|
|
2451
|
+
const adaptiveBudgetCaps = computeAdaptiveJudgePromptBudgetCaps(variants.length);
|
|
2452
|
+
const evaluationContext = resolveJudgeEvaluationContext(taskPrompt ?? "", {
|
|
2453
|
+
evaluationContext: options?.evaluationContext,
|
|
2454
|
+
variants
|
|
2455
|
+
});
|
|
2456
|
+
const spec = parsePromptSpec(taskPrompt ?? "");
|
|
2457
|
+
const fileTargets = evaluationContext.fileTargets.length > 0 ? [...evaluationContext.fileTargets] : resolveAcceptanceCriteriaTargets(taskPrompt ?? "");
|
|
2458
|
+
const evidenceTextByVariant = {};
|
|
2459
|
+
const evidenceCandidatesByVariant = {};
|
|
2460
|
+
const truncationByVariant = {};
|
|
2461
|
+
const promptTelemetryByVariant = {};
|
|
2462
|
+
const worktreePaths = options?.worktreePaths;
|
|
2463
|
+
let contract = evaluationContext.taskContract;
|
|
2464
|
+
if (!contract) {
|
|
2465
|
+
try {
|
|
2466
|
+
contract = extractTaskContract(taskPrompt ?? "", { validateQuotes: false });
|
|
2467
|
+
} catch {
|
|
2468
|
+
}
|
|
2469
|
+
}
|
|
2470
|
+
const hasContractItems = Boolean(
|
|
2471
|
+
contract && (contract.mustHaves.length > 0 || contract.acceptanceCriteria.length > 0)
|
|
2472
|
+
);
|
|
2473
|
+
let hasAnyACComplianceScan = false;
|
|
2474
|
+
for (const variant of variants) {
|
|
2475
|
+
const signals = variant.empiricalSignals;
|
|
2476
|
+
if (!signals) {
|
|
2477
|
+
const missingSignalsBlock = `## Variant: ${variant.variant}
|
|
2478
|
+
|
|
2479
|
+
**Status**: Missing empirical signals (unable to evaluate)
|
|
2480
|
+
`;
|
|
2481
|
+
variantSections.push(missingSignalsBlock);
|
|
2482
|
+
promptTelemetryByVariant[variant.variant] = {
|
|
2483
|
+
factsCard: metricFromChars(0),
|
|
2484
|
+
acComplianceSection: metricFromChars(0),
|
|
2485
|
+
fileInventorySection: metricFromChars(0),
|
|
2486
|
+
metadataSection: metricFromChars(0),
|
|
2487
|
+
deliverableExcerpt: metricFromChars(0),
|
|
2488
|
+
evidenceCandidatesSection: metricFromChars(0),
|
|
2489
|
+
variantBlock: metricFromText(missingSignalsBlock),
|
|
2490
|
+
contentTotal: metricFromChars(0),
|
|
2491
|
+
scaffoldingTotal: metricFromText(missingSignalsBlock)
|
|
2492
|
+
};
|
|
2493
|
+
continue;
|
|
2494
|
+
}
|
|
2495
|
+
const factsCard = formatEmpiricalSignals(signals, variant);
|
|
2496
|
+
const fileInventory = buildFileInventory(variant, {
|
|
2497
|
+
maxFileInventoryFiles: adaptiveBudgetCaps.maxFileInventoryFiles
|
|
2498
|
+
});
|
|
2499
|
+
const { text: deliverableExcerpt, truncation } = buildVariantDeliverableExcerptForJudge(
|
|
2500
|
+
variant,
|
|
2501
|
+
{
|
|
2502
|
+
maxDeliverableExcerptChars: adaptiveBudgetCaps.maxDeliverableExcerptChars,
|
|
2503
|
+
maxDiffExcerptChars: adaptiveBudgetCaps.maxDiffExcerptChars,
|
|
2504
|
+
priorityFiles: fileTargets
|
|
2505
|
+
}
|
|
2506
|
+
);
|
|
2507
|
+
const promptOrderMode = JUDGE_FEATURE_FLAGS.JUDGE_PROFILE;
|
|
2508
|
+
const evidenceCandidateLimit = computePromptOrderEvidenceCandidateLimit(
|
|
2509
|
+
adaptiveBudgetCaps.maxEvidenceCandidates,
|
|
2510
|
+
promptOrderMode
|
|
2511
|
+
);
|
|
2512
|
+
const candidates = buildVariantEvidenceCandidatesForJudge(variant, {
|
|
2513
|
+
maxEvidenceCandidates: evidenceCandidateLimit
|
|
2514
|
+
});
|
|
2515
|
+
evidenceTextByVariant[variant.variant] = deliverableExcerpt.replace(/\r\n/g, "\n");
|
|
2516
|
+
evidenceCandidatesByVariant[variant.variant] = candidates;
|
|
2517
|
+
truncationByVariant[variant.variant] = truncation;
|
|
2518
|
+
const metadataLines = [];
|
|
2519
|
+
const outputNote = buildTruncationNote(
|
|
2520
|
+
truncation.outputFullChars,
|
|
2521
|
+
truncation.outputShownChars,
|
|
2522
|
+
"Output"
|
|
2523
|
+
);
|
|
2524
|
+
const diffNote = buildTruncationNote(
|
|
2525
|
+
truncation.diffFullChars,
|
|
2526
|
+
truncation.diffShownChars,
|
|
2527
|
+
"Diff"
|
|
2528
|
+
);
|
|
2529
|
+
if (outputNote || diffNote) {
|
|
2530
|
+
metadataLines.push(`**Truncation**: ${[outputNote, diffNote].filter(Boolean).join(" | ")}`);
|
|
2531
|
+
}
|
|
2532
|
+
const worktreePath = worktreePaths?.[variant.variant];
|
|
2533
|
+
if (worktreePath) {
|
|
2534
|
+
metadataLines.push(`**Worktree Path**: ${worktreePath}`);
|
|
2535
|
+
}
|
|
2536
|
+
const verificationProvenance = formatVerificationProvenanceForPrompt(variant, {
|
|
2537
|
+
worktreePath
|
|
2538
|
+
});
|
|
2539
|
+
if (verificationProvenance) {
|
|
2540
|
+
metadataLines.push(verificationProvenance);
|
|
2541
|
+
}
|
|
2542
|
+
if (variant.timedOut && variant.partialEdits) {
|
|
2543
|
+
const files = variant.filesChanged ?? variant.worktreeSummary?.filesChanged ?? 0;
|
|
2544
|
+
const added = variant.linesAdded ?? variant.worktreeSummary?.insertions ?? 0;
|
|
2545
|
+
const removed = variant.linesRemoved ?? variant.worktreeSummary?.deletions ?? 0;
|
|
2546
|
+
metadataLines.push(
|
|
2547
|
+
`**Timeout Recovery**: [PARTIAL/INCOMPLETE] partial edits preserved (${files} files, +${added}/-${removed})`
|
|
2548
|
+
);
|
|
2549
|
+
}
|
|
2550
|
+
const metadataSection = metadataLines.length > 0 ? `
|
|
2551
|
+
${metadataLines.join("\n")}
|
|
2552
|
+
` : "";
|
|
2553
|
+
const fileInventorySection = fileInventory ? `
|
|
2554
|
+
${fileInventory}
|
|
2555
|
+
` : "";
|
|
2556
|
+
const acComplianceSection = buildACComplianceSection(contract, variant);
|
|
2557
|
+
if (acComplianceSection) {
|
|
2558
|
+
hasAnyACComplianceScan = true;
|
|
2559
|
+
}
|
|
2560
|
+
const evidenceCandidatesSection = `<<<EVIDENCE CANDIDATES (use IDs E1..E${candidates.length} OR copy/paste the exact line)>>>
|
|
2561
|
+
${candidates.map((c, i) => `E${i + 1}: ${c}`).join("\n")}
|
|
2562
|
+
<<<END EVIDENCE CANDIDATES>>>`;
|
|
2563
|
+
const variantBlock = promptOrderMode === "stable" || promptOrderMode === "shadow" || promptOrderMode === "experimental" ? `## Variant: ${variant.variant}
|
|
2564
|
+
|
|
2565
|
+
${deliverableExcerpt}
|
|
2566
|
+
|
|
2567
|
+
${evidenceCandidatesSection}
|
|
2568
|
+
|
|
2569
|
+
${factsCard}
|
|
2570
|
+
${acComplianceSection ? `${acComplianceSection}
|
|
2571
|
+
` : ""}${fileInventorySection}${metadataSection}
|
|
2572
|
+
` : `## Variant: ${variant.variant}
|
|
2573
|
+
|
|
2574
|
+
${factsCard}
|
|
2575
|
+
${acComplianceSection ? `${acComplianceSection}
|
|
2576
|
+
` : ""}${fileInventorySection}${metadataSection}
|
|
2577
|
+
${deliverableExcerpt}
|
|
2578
|
+
|
|
2579
|
+
${evidenceCandidatesSection}
|
|
2580
|
+
`;
|
|
2581
|
+
variantSections.push(variantBlock);
|
|
2582
|
+
const contentChars = deliverableExcerpt.length + evidenceCandidatesSection.length;
|
|
2583
|
+
promptTelemetryByVariant[variant.variant] = {
|
|
2584
|
+
factsCard: metricFromText(factsCard),
|
|
2585
|
+
acComplianceSection: metricFromText(acComplianceSection ?? ""),
|
|
2586
|
+
fileInventorySection: metricFromText(fileInventorySection),
|
|
2587
|
+
metadataSection: metricFromText(metadataSection),
|
|
2588
|
+
deliverableExcerpt: metricFromText(deliverableExcerpt),
|
|
2589
|
+
evidenceCandidatesSection: metricFromText(evidenceCandidatesSection),
|
|
2590
|
+
variantBlock: metricFromText(variantBlock),
|
|
2591
|
+
contentTotal: metricFromChars(contentChars),
|
|
2592
|
+
scaffoldingTotal: metricFromChars(variantBlock.length - contentChars)
|
|
2593
|
+
};
|
|
2594
|
+
}
|
|
2595
|
+
const verificationIntent = variants[0]?.empiricalSignals?.verificationIntent;
|
|
2596
|
+
const verificationSummary = verificationIntent ? [
|
|
2597
|
+
`**VERIFICATION EXPECTATION**: ${verificationIntent.type} (confidence: ${(verificationIntent.confidence ?? 0).toFixed(2)})`,
|
|
2598
|
+
verificationIntent.evidence?.length ? `Evidence: ${verificationIntent.evidence.join(" | ")}` : null,
|
|
2599
|
+
verificationIntent.overrideSource ? `Override: ${verificationIntent.overrideSource}` : null
|
|
2600
|
+
].filter(Boolean).join("\n") : "";
|
|
2601
|
+
const verificationSection = verificationSummary ? `${verificationSummary}
|
|
2602
|
+
|
|
2603
|
+
` : "";
|
|
2604
|
+
const promptFlags = [
|
|
2605
|
+
"**PROMPT FLAGS**:",
|
|
2606
|
+
` prefersMinimalChange: ${spec.prefersMinimalChange}`,
|
|
2607
|
+
` testsRequested: ${spec.testsRequested}`,
|
|
2608
|
+
` testExecutionRequested: ${spec.testExecutionRequested}`,
|
|
2609
|
+
` testExplicitlyRequired: ${spec.testExplicitlyRequired}`,
|
|
2610
|
+
` isTestIntent: ${spec.isTestIntent}`
|
|
2611
|
+
].join("\n");
|
|
2612
|
+
const resolvedTaskTypeLabel = taskKind ?? evaluationContext.taskTypeResolution.taskType ?? "unknown";
|
|
2613
|
+
const taskTypeSection = `
|
|
2614
|
+
**TASK TYPE**: ${resolvedTaskTypeLabel}`;
|
|
2615
|
+
const rebalanceEnabled = JUDGE_FEATURE_FLAGS.TEST_SIGNAL_REBALANCE === "on";
|
|
2616
|
+
const inferredCodeTask = (taskKind ? /^(implementation|bugfix|refactor|feature|test)$/i.test(taskKind) : false) || !spec.isAnalysisIntent && (spec.fileConstraint.targets.length > 0 || spec.testExecutionRequested || spec.isTestIntent || /\b(code|typescript|javascript|python|test|build|lint|refactor|implementation|function|class|module|bug|fix)\b/i.test(
|
|
2617
|
+
taskPrompt
|
|
2618
|
+
));
|
|
2619
|
+
const testSignalPolicyLine = !rebalanceEnabled ? "- If `testGate.testStatus = failed` with executed tests (`testGate.testsRun > 0`) and `verificationStatus (adjusted, baseline-aware)` is neither `passed` nor `baseline_clean`, correctness should reflect the test failure. If adjusted verification is `baseline_clean`, treat the raw failure as pre-existing unless the diff proves otherwise." : spec.isTestIntent ? "- If `testGate.testStatus = failed` with executed tests (`testGate.testsRun > 0`) on a test-intent task and adjusted verification is neither `passed` nor `baseline_clean`, correctness should reflect the test failure. If adjusted verification is `baseline_clean`, treat the raw failure as pre-existing unless the diff proves otherwise." : spec.testExecutionRequested || spec.testExplicitlyRequired ? "- Task requires tests (per acceptance criteria or explicit request). If variant did not produce or run tests, this is an unmet requirement \u2014 score completeness accordingly. If tests were produced but fail and adjusted verification is neither `passed` nor `baseline_clean`, correctness should reflect that failure." : null;
|
|
2620
|
+
const codeTaskCorrectnessGuidance = inferredCodeTask ? `
|
|
2621
|
+
## Code-Task Correctness Checklist
|
|
2622
|
+
|
|
2623
|
+
- Check for self-referential function calls without a termination condition (infinite recursion risk).
|
|
2624
|
+
- Check for missing/undefined references (imports, variables, functions) and type-level mismatches.
|
|
2625
|
+
- Check for error-handling paths that silently swallow exceptions.
|
|
2626
|
+
- Check for control-flow hazards (unreachable code, switch fallthrough).
|
|
2627
|
+
|
|
2628
|
+
## Correctness Scoring Guidance
|
|
2629
|
+
|
|
2630
|
+
${testSignalPolicyLine ? `${testSignalPolicyLine}
|
|
2631
|
+
` : ""}- If \`buildGate.signal = fail\` and adjusted verification is neither \`passed\` nor \`baseline_clean\`, correctness should reflect the build failure. If adjusted verification is \`baseline_clean\`, treat the raw build failure as pre-existing unless the diff proves otherwise.
|
|
2632
|
+
- If \`Structural Risk Signals\` includes any \`CRITICAL\` finding, correctness should reflect that critical structural risk.
|
|
2633
|
+
- When multiple failure signals are present, weight the most severe.
|
|
2634
|
+
` : "";
|
|
2635
|
+
const inferredBugfixWithParsing = requiresInputSimulationCheck(variants, taskPrompt, taskKind);
|
|
2636
|
+
const inputSimulationGuidance = inferredBugfixWithParsing ? `
|
|
2637
|
+
## Input Simulation Check (required for bugfix + parsing/validation tasks)
|
|
2638
|
+
|
|
2639
|
+
Before scoring Correctness, you MUST trace any validation/parsing logic in each variant's diff against concrete inputs visible in this prompt (variant identifiers, file paths, test data, etc.).
|
|
2640
|
+
|
|
2641
|
+
**Required evidence format** \u2014 for each variant where parsing/validation logic is present, report and return via \`inputSimulationChecks\`:
|
|
2642
|
+
1. **Input**: The concrete string or value used for tracing
|
|
2643
|
+
2. **Code path**: The function/expression being traced
|
|
2644
|
+
3. **Expected result**: What the code should produce for this input
|
|
2645
|
+
4. **Actual result**: What the code actually produces (trace step by step)
|
|
2646
|
+
5. **Verdict**: PASS or FAIL with the specific condition that passes/fails (use SKIP only when no concrete input exists)
|
|
2647
|
+
|
|
2648
|
+
If a variant's parsing logic fails on a real input visible in this prompt, its Correctness score must reflect that failure regardless of how structurally clean the code looks.
|
|
2649
|
+
|
|
2650
|
+
If no concrete inputs are available to trace for a parsing/validation variant, include one \`inputSimulationChecks\` item with verdict=SKIP and note "No concrete inputs available for simulation."
|
|
2651
|
+
` : "";
|
|
2652
|
+
const inputSimulationExplainabilityRequirement = inferredBugfixWithParsing ? "\n7. **inputSimulationChecks**: Required for variants with parsing/validation changes. Provide at least one check object. Use verdict=SKIP only with an explicit note when no concrete input exists." : "";
|
|
2653
|
+
const acComplianceScoringGuidance = hasContractItems && hasAnyACComplianceScan ? `
|
|
2654
|
+
## AC Compliance Guidance
|
|
2655
|
+
|
|
2656
|
+
Treat the AC Compliance Scan as heuristic evidence only.
|
|
2657
|
+
|
|
2658
|
+
First determine whether the variant actually satisfies the requirement based on the strongest available evidence.
|
|
2659
|
+
- If the requirement is satisfied through equivalent implementation or other direct evidence, cite that evidence and score normally.
|
|
2660
|
+
- Only treat an AC Compliance miss as negative delivery evidence when the requirement appears genuinely unmet.
|
|
2661
|
+
|
|
2662
|
+
Guidance:
|
|
2663
|
+
- A [NOT FOUND] [MUST] result is strong heuristic evidence of incompleteness, but it is not proof by itself.
|
|
2664
|
+
- A [NOT FOUND] [AC] or [PARTIAL] [AC] result should generally lower Delivery unless stronger evidence shows the requirement was in fact satisfied.
|
|
2665
|
+
- When a requirement appears unmet, include it in \`delivery.missedRequirements\`.
|
|
2666
|
+
- Do not let lexical or structural proxies outweigh stronger contradictory outcome evidence.
|
|
2667
|
+
- Implementation-detail boundary: internal helper names, object property names, constants, internal/private data-shape choices, and naming terminology are NOT acceptance requirements unless the task text, existing public API, or existing tests name that exact contract. If two variants satisfy the same user-visible behavior with different internal field names, do not mark either one as a requirement failure solely for that naming difference.
|
|
2668
|
+
|
|
2669
|
+
Score guidance:
|
|
2670
|
+
- If a MUST requirement appears genuinely unmet, Delivery should generally not exceed 60.
|
|
2671
|
+
- If an AC item appears genuinely unmet or only partially met, Delivery should generally not exceed 75.
|
|
2672
|
+
` : "";
|
|
2673
|
+
const finalScoreAuthorityNote = `
|
|
2674
|
+
## Final Score Authority
|
|
2675
|
+
|
|
2676
|
+
Your \`os\` (overall score) is authoritative. The reporting formula weights ${THREE_BUCKET_RUBRIC_PERCENT_TEXT}. If your \`os\` diverges from that formula, explain why in your rationale.
|
|
2677
|
+
`;
|
|
2678
|
+
const framedTaskPrompt = frameUntrustedJudgeTaskText(taskPrompt ?? "");
|
|
2679
|
+
const prompt = `You are evaluating ${variants.length} AI-generated variant(s) for the following task:
|
|
2680
|
+
|
|
2681
|
+
**TASK**: ${framedTaskPrompt}
|
|
2682
|
+
${taskTypeSection}
|
|
2683
|
+
|
|
2684
|
+
${verificationSection}${promptFlags}
|
|
2685
|
+
|
|
2686
|
+
---
|
|
2687
|
+
|
|
2688
|
+
# Empirical Facts for Each Variant
|
|
2689
|
+
|
|
2690
|
+
${variantSections.join("\n---\n\n")}
|
|
2691
|
+
|
|
2692
|
+
---
|
|
2693
|
+
|
|
2694
|
+
# Text Evidence Requirements (REQUIRED)
|
|
2695
|
+
|
|
2696
|
+
You MUST include short supporting evidence for each variant, grounded in the deliverables above.
|
|
2697
|
+
|
|
2698
|
+
Evidence format (choose one):
|
|
2699
|
+
- **Preferred**: Use evidence IDs from the **EVIDENCE CANDIDATES** list (e.g., "E3")
|
|
2700
|
+
- Or: Copy/paste a short quote that is a verbatim substring from the variant's **DELIVERABLE EXCERPT**
|
|
2701
|
+
|
|
2702
|
+
Use verbatim text from excerpts. The evaluator validates that evidence matches the provided text exactly.
|
|
2703
|
+
|
|
2704
|
+
**CRITICAL CONSTRAINTS (checked after generation)**:
|
|
2705
|
+
- **supportingQuotes**: Provide EXACTLY 1-2 items per variant.
|
|
2706
|
+
- Quotes must be **exact substrings** from the excerpt.
|
|
2707
|
+
- Keep quotes short (prefer <= 200 chars).
|
|
2708
|
+
- Use quotes to support your biggest claim(s) for the variant.
|
|
2709
|
+
- If you cannot find supporting quotes for a claim, you MUST lower confidence and describe the uncertainty.
|
|
2710
|
+
- Missing or invalid quotes are flagged during judge validation and reduce trust in the response quality.
|
|
2711
|
+
|
|
2712
|
+
---
|
|
2713
|
+
|
|
2714
|
+
# Anti-Verbosity Policy (ENFORCED)
|
|
2715
|
+
|
|
2716
|
+
- **Length is NOT quality**: A longer explanation or more code does NOT merit a higher score
|
|
2717
|
+
- **Evaluate substance only**: Score strictly on requirements met, correctness evidence, and code clarity
|
|
2718
|
+
- **Concise can be better**: A concise solution that fully meets requirements is EQUAL OR SUPERIOR to verbose output
|
|
2719
|
+
- **Bias self-check**: If you prefer a variant for "more detail," confirm that detail was explicitly required
|
|
2720
|
+
- **Score willingness**: Be equally willing to assign low scores when the evidence supports it. Do not use fixed host-side score caps.
|
|
2721
|
+
- **Independent evaluation**: Evaluate each variant's strengths and weaknesses independently before making comparisons between variants.
|
|
2722
|
+
|
|
2723
|
+
---
|
|
2724
|
+
|
|
2725
|
+
# Deterministic Signal Guidance
|
|
2726
|
+
|
|
2727
|
+
**Consider the following when scoring/ranking variants:**
|
|
2728
|
+
|
|
2729
|
+
1. **Deterministic Signals Are Evidence, Not Filters**:
|
|
2730
|
+
- Build, lint, test, execution, and deliverable signals are factual evidence for your judgment.
|
|
2731
|
+
- Do not mark variants ineligible because of deterministic gates.
|
|
2732
|
+
- You may choose a variant with failing verification if, after weighing all evidence, it is still the best answer.
|
|
2733
|
+
- Explain clearly when the winner has failing or missing verification evidence.
|
|
2734
|
+
|
|
2735
|
+
2. **Quality and Correctness Are Holistic**:
|
|
2736
|
+
- Use the 3-bucket rubric to determine ranking across all variants.
|
|
2737
|
+
- Higher quality, better delivery, and correctness evidence should drive the final decision.
|
|
2738
|
+
|
|
2739
|
+
3. **Cite Gate Artifacts (REQUIRED)**:
|
|
2740
|
+
- When making claims about builds, tests, or prompt constraints, you MUST cite recorded artifacts:
|
|
2741
|
+
- Build status: Reference \`buildGate.signal\` or \`buildGate.exitCode\`
|
|
2742
|
+
- Test execution: Reference \`testGate.testStatus\`, \`testGate.testsRun\`, or \`testGate.exitCode\`
|
|
2743
|
+
- Execution success: Reference \`execution.success\` or \`execution.failureType\`
|
|
2744
|
+
- Deliverable constraints: Reference \`deliverableTargets\` / \`deliverableSatisfied\` (when present)
|
|
2745
|
+
- Example: "Build passed (buildGate.signal: healthy, exitCode: 0)"
|
|
2746
|
+
- Example: "Tests failed (testGate.testStatus: failed, exitCode: 1, failed: 2)"
|
|
2747
|
+
|
|
2748
|
+
4. **Deliverable Compliance**: When deliverable targets are specified (see
|
|
2749
|
+
\`deliverableTargets\` in empirical signals), treat missing or partial deliverables
|
|
2750
|
+
as evidence. Do not apply fixed delivery caps; judge the comparative quality and explain the impact.
|
|
2751
|
+
|
|
2752
|
+
5. **Insufficient Implementation**: Treat planning-only, tests-only, missing-change, and timed-out partial outputs as evidence. Do not apply fixed caps or automatic zero scores; assign the score your judgment supports and explain the tradeoff.
|
|
2753
|
+
|
|
2754
|
+
6. **Winner Authority**:
|
|
2755
|
+
- You decide whether there is a winner.
|
|
2756
|
+
- If every variant has problems, still choose the best one unless you determine no variant should win.
|
|
2757
|
+
|
|
2758
|
+
7. **AC Compliance Guidance**:
|
|
2759
|
+
- First determine whether the requirement is actually satisfied based on the strongest available evidence.
|
|
2760
|
+
- If satisfied through equivalent implementation or direct evidence, cite that evidence and score normally.
|
|
2761
|
+
- A [NOT FOUND] [MUST] result is strong heuristic evidence of incompleteness; if the requirement appears genuinely unmet, Delivery should generally not exceed 60.
|
|
2762
|
+
- A [NOT FOUND] [AC] or [PARTIAL] [AC] result should generally lower Delivery unless stronger evidence shows the requirement was in fact satisfied.
|
|
2763
|
+
- When a requirement appears unmet, include it in \`delivery.missedRequirements\`.
|
|
2764
|
+
${JUDGE_FEATURE_FLAGS.VARIANT_DIFFERENTIATION_SIGNALS === "on" ? `
|
|
2765
|
+
8. **Score Differentiation (REQUIRED)**:
|
|
2766
|
+
- When variants have different diffs or different empirical signals, you MUST either:
|
|
2767
|
+
(a) Cite a concrete differentiator in at least one bucket and reflect it in scoring, OR
|
|
2768
|
+
(b) Provide an equivalence justification: state what specific evidence you examined,
|
|
2769
|
+
confirm the differences are irrelevant to all three buckets, and explain why
|
|
2770
|
+
in \`singleBiggestReason\`.
|
|
2771
|
+
- Identical scores for non-identical implementations without explicit equivalence
|
|
2772
|
+
justification are treated as insufficient analysis.
|
|
2773
|
+
` : ""}
|
|
2774
|
+
---
|
|
2775
|
+
|
|
2776
|
+
# Evaluation Instructions
|
|
2777
|
+
|
|
2778
|
+
Evaluate each variant using the **3-bucket rubric** (Delivery 50%, Correctness 30%, Quality 20%).
|
|
2779
|
+
${worktreePaths && Object.keys(worktreePaths).length > 0 ? `
|
|
2780
|
+
## File Access
|
|
2781
|
+
|
|
2782
|
+
You have read-only access to each variant's worktree via the paths listed above.
|
|
2783
|
+
Use Read, Glob, and Grep tools when truncation metadata shows significant evidence
|
|
2784
|
+
was hidden (e.g., less than 50% visible). Start with the provided excerpts; deep-dive
|
|
2785
|
+
into the worktree only when the excerpts are insufficient to score confidently.
|
|
2786
|
+
Do NOT modify any files.
|
|
2787
|
+
` : ""}
|
|
2788
|
+
## Bucket Definitions (adapt based on task type)
|
|
2789
|
+
|
|
2790
|
+
### For implementation/bugfix/refactor tasks:
|
|
2791
|
+
- **Delivery (50%)**: Did the work - requirements met, completion
|
|
2792
|
+
- **Correctness (30%)**: Likely correct behavior - edge cases, test signals, risk assessment
|
|
2793
|
+
- **Quality (20%)**: Readability, maintainability, code cleanliness
|
|
2794
|
+
|
|
2795
|
+
### For analysis/design/planning tasks:
|
|
2796
|
+
- **Delivery (50%)**: Scope coverage and completeness of response
|
|
2797
|
+
- **Correctness (30%)**: Reasoning consistency, factual grounding
|
|
2798
|
+
- **Quality (20%)**: Clarity, structure, actionability
|
|
2799
|
+
|
|
2800
|
+
${codeTaskCorrectnessGuidance}
|
|
2801
|
+
${inputSimulationGuidance}
|
|
2802
|
+
${acComplianceScoringGuidance}
|
|
2803
|
+
|
|
2804
|
+
## Proxy vs Outcome Evidence
|
|
2805
|
+
|
|
2806
|
+
Distinguish between proxy evidence and outcome evidence.
|
|
2807
|
+
|
|
2808
|
+
- Proxy evidence includes naming alignment, documentation updates, structural resemblance to the requested change, keyword matches, and configuration patterns that look directionally correct.
|
|
2809
|
+
- Outcome evidence includes direct proof that the requested result was achieved: successful behavior, required targets actually satisfied, failing condition eliminated, or explicit verification evidence.
|
|
2810
|
+
|
|
2811
|
+
Outcome evidence is stronger than proxy evidence.
|
|
2812
|
+
|
|
2813
|
+
Important:
|
|
2814
|
+
- Multiple corroborating proxies can still be jointly wrong when none directly measures the requested outcome.
|
|
2815
|
+
- Do not treat documentation quality, naming alignment, or requirement-word overlap as proof that the task outcome was achieved.
|
|
2816
|
+
- When proxy evidence suggests completion but outcome evidence is missing or contradictory, discount the proxies and score accordingly.
|
|
2817
|
+
- When two variants achieve the same outcome, proxy evidence such as clarity and documentation may still differentiate quality and delivery.
|
|
2818
|
+
- Internal implementation details are not task requirements unless explicitly named by the task, existing public API, or existing tests. Do not turn one variant's private field names, constants, helper names, or internal/private data-shape choices into requirements for other variants.
|
|
2819
|
+
- When task-appropriate deterministic verification passes and the evidence shows the requested requirements implemented, do not invent a weakness from excerpt visibility alone (for example, "not explicitly visible in diff excerpt"). Penalize only concrete task-relevant defects, unmet requirements, contradictory evidence, or real verification gaps.
|
|
2820
|
+
|
|
2821
|
+
${finalScoreAuthorityNote}
|
|
2822
|
+
## Scoring Guidelines
|
|
2823
|
+
|
|
2824
|
+
**Delivery Scoring**:
|
|
2825
|
+
- 90-100: All requirements met, task fully completed
|
|
2826
|
+
- 70-89: Most requirements met, good completion
|
|
2827
|
+
- 50-69: Partial requirements met, some gaps
|
|
2828
|
+
- 30-49: Minimal delivery, many gaps
|
|
2829
|
+
- 0-29: Execution failed or no meaningful output
|
|
2830
|
+
|
|
2831
|
+
**Correctness Scoring**:
|
|
2832
|
+
- 90-100: High confidence in correctness, strong test coverage (testQualityLevel: proves_cli)
|
|
2833
|
+
- 70-89: Good correctness signals, adequate tests (testQualityLevel: validates_logic)
|
|
2834
|
+
- 50-69: Some correctness concerns, weak test signals
|
|
2835
|
+
- 30-49: Significant correctness risks
|
|
2836
|
+
- 0-29: Major correctness failures or execution errors
|
|
2837
|
+
|
|
2838
|
+
**Quality Scoring**:
|
|
2839
|
+
- 90-100: Excellent code quality, clear patterns
|
|
2840
|
+
- 70-89: Good quality, minor issues
|
|
2841
|
+
- 50-69: Acceptable quality, some maintainability concerns
|
|
2842
|
+
- 30-49: Quality issues detected (magic numbers, mutations, poor structure)
|
|
2843
|
+
- 0-29: Poor quality
|
|
2844
|
+
|
|
2845
|
+
**Evaluation context notes**:
|
|
2846
|
+
- **Change size neutrality**: Judge by requirements met and correctness evidence, not change volume. Fewer lines or more lines are irrelevant unless the prompt explicitly requests minimal/surgical change or the task clearly implies a tiny correction (e.g., a single comment/doc/typo fix). In those cases, unnecessary extra scope or new files/APIs should lower Delivery.
|
|
2847
|
+
- **For analysis/design/planning tasks**: Evaluate scope and reasoning quality; code changes are not expected.
|
|
2848
|
+
- **Timed-out partials**: If \`execution.timedOut = true\` and \`execution.partialEdits = true\`, evaluate the marked \`[PARTIAL/INCOMPLETE]\` diff and award partial credit based on delivered scope. Do not auto-score Delivery as zero solely because of timeout.
|
|
2849
|
+
- **Tests neutrality**: If the prompt does not request tests (testsRequested/testExecutionRequested/isTestIntent are all false), treat test additions as neutral \u2014 neither reward nor penalize.
|
|
2850
|
+
- **Test alignment edits**: Do not assume changed test expectations are masking failures. For test stabilization or source/test contract repair, stale expectations should be updated when current source behavior, provider registry defaults, or verified failure output supports the change. Penalize test edits as masking only when stronger source or outcome evidence contradicts the new expectation.
|
|
2851
|
+
- **Test quality score (0-15)** is only relevant when tests are part of the task (testsRequested or isTestIntent).
|
|
2852
|
+
- **Test execution results** are only relevant when running tests is part of the task (testExecutionRequested or isTestIntent).
|
|
2853
|
+
- **Tests vs task quality**: If tests are requested as a deliverable (testsRequested/isTestIntent), check whether test files were actually changed. Prioritize correctness of the primary task first, and use test pass/fail as a tie-breaker when candidates are truly close.
|
|
2854
|
+
- **Top risk severity hints**: breaking changes \u2192 major, mutations \u2192 minor (unless clearly dangerous).
|
|
2855
|
+
|
|
2856
|
+
## Score Calibration
|
|
2857
|
+
|
|
2858
|
+
Apply to each bucket independently:
|
|
2859
|
+
90-100: Flawless execution \u2014 no defects, no missed scope, production-ready.
|
|
2860
|
+
70-89: Solid implementation \u2014 achieves the goal with 1-2 minor gaps.
|
|
2861
|
+
50-69: Partial success \u2014 core intent addressed but meaningful gaps remain.
|
|
2862
|
+
30-49: Significant problems \u2014 introduces new issues or misses major requirements.
|
|
2863
|
+
0-29: Fundamentally broken or did not attempt the task.
|
|
2864
|
+
Use the full range. A variant scoring 50 in one bucket can score 90 in another.
|
|
2865
|
+
|
|
2866
|
+
## Confidence Calibration
|
|
2867
|
+
|
|
2868
|
+
Use these anchors when determining your confidence score:
|
|
2869
|
+
- **95-100**: Unambiguous winner; nearly any evaluator would agree
|
|
2870
|
+
- **85-94**: Strong structural evidence (e.g., one variant produced deliverables while the other did not, one variant is clearly off-task, one variant's core feature is non-functional or has critical defects while the other works, or one variant omits required targets); minor residual uncertainty
|
|
2871
|
+
- **70-84**: Clear lean toward winner; evidence supports the ranking but some dimensions are close
|
|
2872
|
+
- **50-69**: Meaningful uncertainty; winner is plausible but alternative ranking is defensible
|
|
2873
|
+
- **40-49**: Close call; multiple plausible winners with thin evidence separation
|
|
2874
|
+
- **<40**: Insufficient evidence to determine a clear winner
|
|
2875
|
+
|
|
2876
|
+
Confidence MUST reflect actual evidence quality, not just score separation. However, when structural evidence is decisive (deliverables produced vs none, on-task vs off-task, passing tests vs failing, core feature functional vs non-functional, required targets covered vs omitted), confidence should be at least 85 even if individual evidence excerpts are thin.
|
|
2877
|
+
|
|
2878
|
+
## Score Spread Guidance
|
|
2879
|
+
|
|
2880
|
+
- When variants have different implementations, their weighted totals SHOULD differ by at least 3 points. If within a 3-point window, provide equivalence justification in summary.rankingRationale.
|
|
2881
|
+
- Use the full 0-100 range per bucket. Build failures \u2192 correctness below 50. All requirements met with clean build \u2192 delivery above 80.
|
|
2882
|
+
- If all scores cluster within a 10-point window across all buckets, flag in summary.rankingRationale.
|
|
2883
|
+
|
|
2884
|
+
## Explainability Requirements (ENFORCED)
|
|
2885
|
+
|
|
2886
|
+
For each variant, you MUST provide:
|
|
2887
|
+
1. **deliveredWell**: 1-3 items OR ["No notable deliverables"]
|
|
2888
|
+
2. **correctnessStrengths**: 1-3 items OR ["No correctness strengths identified"]
|
|
2889
|
+
3. **qualityHighlights**: 1-2 items (required if quality.score > 70)
|
|
2890
|
+
4. **qualityIssues**: 1-2 items (required if quality.score < 85)
|
|
2891
|
+
5. **missedRequirements**: may be empty ONLY if delivery.score >= 95
|
|
2892
|
+
|
|
2893
|
+
**Polarity rule**: deliveredWell, correctnessStrengths, and qualityHighlights must contain ONLY positive observations. Do not append caveats, limitations, or negative findings to entries in these fields. Route defects and risks to missedRequirements, topRisk, or qualityIssues respectively.
|
|
2894
|
+
6. **singleBiggestReason**: MUST be a structured evidence-bound object with three fields:
|
|
2895
|
+
- **field**: Must reference an empiricalSignals field (changedFiles, diffStats, execution, testMetrics, codePatterns, buildGate, testGate, etc.)
|
|
2896
|
+
- **value**: The actual observed value from empiricalSignals (e.g., 'none', 3, or object)
|
|
2897
|
+
- **implication**: Clear explanation of what this evidence means for the score
|
|
2898
|
+
- **Gate artifact citation**: When referencing gates, cite specific artifacts like exitCode, signal, stdout/stderr snippets
|
|
2899
|
+
${inputSimulationExplainabilityRequirement}
|
|
2900
|
+
|
|
2901
|
+
When providing \`summary.rankingRationale\`, keep it concise (under 600 characters). Focus on the single most decisive factor and cite at least one evidence ID (E1, E2...) or file path from the evidence candidates. Detailed analysis belongs in per-variant \`singleBiggestReason\` fields.
|
|
2902
|
+
|
|
2903
|
+
**REQUIRED**: For multi-variant evaluations, compute \`os\` (overallScore) per variant as: ${THREE_BUCKET_RUBRIC_FORMULA_TEXT}. Include \`summary.ranking\` with all variant IDs ordered by \`os\` descending. \`summary.ranking[0]\` is the winner unless you explicitly return \`"winner": null\`; that null declares that no variant should win while preserving the comparative ranking. Missing \`os\` or \`ranking\` causes schema failure.
|
|
2904
|
+
|
|
2905
|
+
---
|
|
2906
|
+
|
|
2907
|
+
# Output Format
|
|
2908
|
+
|
|
2909
|
+
Return ONLY valid JSON. No markdown code fences, no prose, no explanatory text.
|
|
2910
|
+
|
|
2911
|
+
Field rules: every "score" and "os" is an integer 0-100; "confidence.value" is a number
|
|
2912
|
+
between 0.0 and 1.0; \`a|b|c\` in a string value means pick exactly one of those values.
|
|
2913
|
+
The example below is valid, parseable JSON \u2014 emit the same shape with your own values.
|
|
2914
|
+
|
|
2915
|
+
Example:
|
|
2916
|
+
|
|
2917
|
+
{
|
|
2918
|
+
"taskType": "implementation",
|
|
2919
|
+
"variants": [
|
|
2920
|
+
{
|
|
2921
|
+
"variant": "variant-name",
|
|
2922
|
+
"delivery": {
|
|
2923
|
+
"score": 78,
|
|
2924
|
+
"missedRequirements": ["requirement 1", "requirement 2"],
|
|
2925
|
+
"deliveredWell": ["positive 1", "positive 2", "positive 3"]
|
|
2926
|
+
},
|
|
2927
|
+
"correctness": {
|
|
2928
|
+
"score": 72,
|
|
2929
|
+
"topRisk": {
|
|
2930
|
+
"summary": "brief risk description",
|
|
2931
|
+
"severity": "minor",
|
|
2932
|
+
"details": "optional detailed explanation"
|
|
2933
|
+
},
|
|
2934
|
+
"correctnessStrengths": ["strength 1", "strength 2", "strength 3"]
|
|
2935
|
+
},
|
|
2936
|
+
"quality": {
|
|
2937
|
+
"score": 69,
|
|
2938
|
+
"qualityHighlights": ["highlight 1", "highlight 2"],
|
|
2939
|
+
"qualityIssues": ["issue 1", "issue 2"],
|
|
2940
|
+
"offTargetEdits": {
|
|
2941
|
+
"changedFilesNotInTargets": ["file1.ts", "file2.ts"],
|
|
2942
|
+
"note": "optional justification"
|
|
2943
|
+
}
|
|
2944
|
+
},
|
|
2945
|
+
"singleBiggestReason": {
|
|
2946
|
+
"field": "execution.success",
|
|
2947
|
+
"value": true,
|
|
2948
|
+
"implication": "Execution completed successfully with evidence-backed deliverables"
|
|
2949
|
+
},
|
|
2950
|
+
"confidence": {
|
|
2951
|
+
"value": 0.75,
|
|
2952
|
+
"reason": [
|
|
2953
|
+
"evidence supporting confidence level",
|
|
2954
|
+
"factors creating uncertainty (if any)"
|
|
2955
|
+
]
|
|
2956
|
+
},
|
|
2957
|
+
"inputSimulationChecks": [
|
|
2958
|
+
{
|
|
2959
|
+
"input": "cursor:auto:innovative:local",
|
|
2960
|
+
"codePath": "isCanonicalVariantKey()",
|
|
2961
|
+
"expectedResult": "true (valid canonical key)",
|
|
2962
|
+
"actualResult": "false when parts.length === 4 check rejects model containing ':'",
|
|
2963
|
+
"verdict": "FAIL",
|
|
2964
|
+
"note": "Use verdict=SKIP with note only when no concrete input is available."
|
|
2965
|
+
}
|
|
2966
|
+
],
|
|
2967
|
+
"supportingQuotes": ["E1", "E2"],
|
|
2968
|
+
"os": 79
|
|
2969
|
+
}
|
|
2970
|
+
],
|
|
2971
|
+
"summary": {
|
|
2972
|
+
"ranking": ["variant-best", "variant-worst"],
|
|
2973
|
+
"rankingRationale": "concrete reason citing empiricalSignals",
|
|
2974
|
+
"confidence": 0.82,
|
|
2975
|
+
"confidenceRationale": ["strong evidence from tests/build signals", "minor uncertainty remains"],
|
|
2976
|
+
"top2SeparationRationale": "why top score beat runner-up"
|
|
2977
|
+
}
|
|
2978
|
+
}
|
|
2979
|
+
|
|
2980
|
+
**CRITICAL**: Return ONLY the JSON object above (with your values filled in). No markdown fences, no explanatory text.
|
|
2981
|
+
Set \`"winner": null\` only when no variant should win. Otherwise omit it and summary.ranking[0] wins.
|
|
2982
|
+
`;
|
|
2983
|
+
const variantBlocksText = variantSections.join("\n---\n\n");
|
|
2984
|
+
const diffContentChars = Object.values(promptTelemetryByVariant).reduce(
|
|
2985
|
+
(sum, telemetry) => sum + telemetry.deliverableExcerpt.chars,
|
|
2986
|
+
0
|
|
2987
|
+
);
|
|
2988
|
+
const evidenceCandidatesChars = Object.values(promptTelemetryByVariant).reduce(
|
|
2989
|
+
(sum, telemetry) => sum + telemetry.evidenceCandidatesSection.chars,
|
|
2990
|
+
0
|
|
2991
|
+
);
|
|
2992
|
+
const variantBlocksChars = variantBlocksText.length;
|
|
2993
|
+
const totalChars = prompt.length;
|
|
2994
|
+
const taskAndInstructionsChars = Math.max(0, totalChars - variantBlocksChars);
|
|
2995
|
+
const scaffoldingChars = Math.max(0, totalChars - diffContentChars - evidenceCandidatesChars);
|
|
2996
|
+
const promptTelemetry = {
|
|
2997
|
+
estimator: "chars_div_4",
|
|
2998
|
+
path: "3-bucket",
|
|
2999
|
+
variants: promptTelemetryByVariant,
|
|
3000
|
+
sections: {
|
|
3001
|
+
taskAndInstructions: metricFromChars(taskAndInstructionsChars),
|
|
3002
|
+
variantBlocks: metricFromChars(variantBlocksChars),
|
|
3003
|
+
diffContent: metricFromChars(diffContentChars),
|
|
3004
|
+
evidenceCandidates: metricFromChars(evidenceCandidatesChars),
|
|
3005
|
+
scaffoldingTotal: metricFromChars(scaffoldingChars),
|
|
3006
|
+
totalPrompt: metricFromChars(totalChars)
|
|
3007
|
+
}
|
|
3008
|
+
};
|
|
3009
|
+
return {
|
|
3010
|
+
prompt,
|
|
3011
|
+
evidenceTextByVariant,
|
|
3012
|
+
evidenceCandidatesByVariant,
|
|
3013
|
+
truncationByVariant,
|
|
3014
|
+
promptTelemetry,
|
|
3015
|
+
evaluationContext
|
|
3016
|
+
};
|
|
3017
|
+
}
|
|
3018
|
+
|
|
3019
|
+
// src/core/judge/anonymization.ts
|
|
3020
|
+
import { createHash } from "crypto";
|
|
3021
|
+
var SeededRandom = class {
|
|
3022
|
+
constructor(seed) {
|
|
3023
|
+
this.seed = seed;
|
|
3024
|
+
}
|
|
3025
|
+
seed;
|
|
3026
|
+
counter = 0;
|
|
3027
|
+
/**
|
|
3028
|
+
* Generate next random value between 0 and 1
|
|
3029
|
+
*/
|
|
3030
|
+
next() {
|
|
3031
|
+
const hash = createHash("sha256");
|
|
3032
|
+
hash.update(`${this.seed}-${this.counter}`);
|
|
3033
|
+
this.counter++;
|
|
3034
|
+
const hex = hash.digest("hex");
|
|
3035
|
+
const value = parseInt(hex.substring(0, 8), 16);
|
|
3036
|
+
return value / 4294967295;
|
|
3037
|
+
}
|
|
3038
|
+
};
|
|
3039
|
+
function createAnonymizationSeed(taskId, runId) {
|
|
3040
|
+
const seedInput = runId ? `${taskId}:${runId}` : taskId;
|
|
3041
|
+
return createHash("sha256").update(seedInput).digest("hex").slice(0, 16);
|
|
3042
|
+
}
|
|
3043
|
+
function createFallbackAnonymizationSeed(prompt, runId) {
|
|
3044
|
+
const seedInput = runId ? `prompt:${prompt}:${runId}` : `prompt:${prompt}`;
|
|
3045
|
+
return createHash("sha256").update(seedInput).digest("hex").slice(0, 16);
|
|
3046
|
+
}
|
|
3047
|
+
function seededShuffle(array, seed) {
|
|
3048
|
+
const result = [...array];
|
|
3049
|
+
const rng = new SeededRandom(seed);
|
|
3050
|
+
for (let i = result.length - 1; i > 0; i--) {
|
|
3051
|
+
const j = Math.floor(rng.next() * (i + 1));
|
|
3052
|
+
const tmp = result[i];
|
|
3053
|
+
result[i] = result[j];
|
|
3054
|
+
result[j] = tmp;
|
|
3055
|
+
}
|
|
3056
|
+
return result;
|
|
3057
|
+
}
|
|
3058
|
+
function generateAnonymousLabel(index) {
|
|
3059
|
+
if (index < 26) {
|
|
3060
|
+
return `Solution ${String.fromCharCode(65 + index)}`;
|
|
3061
|
+
}
|
|
3062
|
+
return `Solution ${index + 1}`;
|
|
3063
|
+
}
|
|
3064
|
+
function anonymizeVariants(variants, taskId, runId) {
|
|
3065
|
+
if (variants.length === 0) {
|
|
3066
|
+
return {
|
|
3067
|
+
shuffledVariants: [],
|
|
3068
|
+
originalToShuffled: /* @__PURE__ */ new Map(),
|
|
3069
|
+
shuffledToOriginal: /* @__PURE__ */ new Map(),
|
|
3070
|
+
variantToLabel: /* @__PURE__ */ new Map(),
|
|
3071
|
+
labelToVariant: /* @__PURE__ */ new Map(),
|
|
3072
|
+
seed: createAnonymizationSeed(taskId, runId)
|
|
3073
|
+
};
|
|
3074
|
+
}
|
|
3075
|
+
const seed = createAnonymizationSeed(taskId, runId);
|
|
3076
|
+
const indices = variants.map((_, i) => i);
|
|
3077
|
+
const shuffledIndices = seededShuffle(indices, seed);
|
|
3078
|
+
const shuffledVariants = shuffledIndices.map((originalIndex) => variants[originalIndex]);
|
|
3079
|
+
const originalToShuffled = /* @__PURE__ */ new Map();
|
|
3080
|
+
const shuffledToOriginal = /* @__PURE__ */ new Map();
|
|
3081
|
+
const variantToLabel = /* @__PURE__ */ new Map();
|
|
3082
|
+
const labelToVariant = /* @__PURE__ */ new Map();
|
|
3083
|
+
for (let shuffledIndex = 0; shuffledIndex < shuffledIndices.length; shuffledIndex++) {
|
|
3084
|
+
const originalIndex = shuffledIndices[shuffledIndex];
|
|
3085
|
+
originalToShuffled.set(originalIndex, shuffledIndex);
|
|
3086
|
+
shuffledToOriginal.set(shuffledIndex, originalIndex);
|
|
3087
|
+
const variant = variants[originalIndex];
|
|
3088
|
+
const label = generateAnonymousLabel(shuffledIndex);
|
|
3089
|
+
if (variantToLabel.has(variant.variant)) {
|
|
3090
|
+
throw new Error(
|
|
3091
|
+
`Anonymization identity collision: duplicate variant key "${variant.variant}". Variant identifiers must be unique before anonymization; a collision would cause label\u2194variant misattribution during de-anonymization.`
|
|
3092
|
+
);
|
|
3093
|
+
}
|
|
3094
|
+
variantToLabel.set(variant.variant, label);
|
|
3095
|
+
labelToVariant.set(label, variant.variant);
|
|
3096
|
+
}
|
|
3097
|
+
return {
|
|
3098
|
+
shuffledVariants,
|
|
3099
|
+
originalToShuffled,
|
|
3100
|
+
shuffledToOriginal,
|
|
3101
|
+
variantToLabel,
|
|
3102
|
+
labelToVariant,
|
|
3103
|
+
seed
|
|
3104
|
+
};
|
|
3105
|
+
}
|
|
3106
|
+
function restoreOriginalOrder(evaluations, mapping) {
|
|
3107
|
+
if (evaluations.length > 0 && evaluations.length !== mapping.size) {
|
|
3108
|
+
throw new Error(
|
|
3109
|
+
`restoreOriginalOrder: evaluation count mismatch \u2014 ${evaluations.length} evaluation(s) returned for ${mapping.size} variant(s). The judge returned a partial evaluation set; restoring would drop or duplicate a variant silently.`
|
|
3110
|
+
);
|
|
3111
|
+
}
|
|
3112
|
+
const restored = new Array(evaluations.length);
|
|
3113
|
+
for (let shuffledIndex = 0; shuffledIndex < evaluations.length; shuffledIndex++) {
|
|
3114
|
+
const originalIndex = mapping.get(shuffledIndex);
|
|
3115
|
+
if (originalIndex === void 0) {
|
|
3116
|
+
throw new Error(
|
|
3117
|
+
`restoreOriginalOrder: no original-index mapping for shuffled index ${shuffledIndex}. The shuffled\u2192original mapping is incomplete relative to the evaluations being restored.`
|
|
3118
|
+
);
|
|
3119
|
+
}
|
|
3120
|
+
if (originalIndex >= evaluations.length) {
|
|
3121
|
+
throw new Error(
|
|
3122
|
+
`restoreOriginalOrder: evaluation count mismatch \u2014 shuffled index ${shuffledIndex} maps to original index ${originalIndex}, which is out of bounds for ${evaluations.length} evaluation(s). The judge returned fewer evaluations than variants; a variant would be dropped silently.`
|
|
3123
|
+
);
|
|
3124
|
+
}
|
|
3125
|
+
restored[originalIndex] = evaluations[shuffledIndex];
|
|
3126
|
+
}
|
|
3127
|
+
return restored;
|
|
3128
|
+
}
|
|
3129
|
+
function resolveAnonymizationRunId(options) {
|
|
3130
|
+
const runIdFromOptions = options.runId;
|
|
3131
|
+
if (typeof runIdFromOptions === "string" && runIdFromOptions.trim().length > 0) {
|
|
3132
|
+
return runIdFromOptions.trim();
|
|
3133
|
+
}
|
|
3134
|
+
const labRunId = readEnv(POETIC_LAB_RUN_ID).value;
|
|
3135
|
+
if (typeof labRunId === "string" && labRunId.trim().length > 0) {
|
|
3136
|
+
return labRunId.trim();
|
|
3137
|
+
}
|
|
3138
|
+
return void 0;
|
|
3139
|
+
}
|
|
3140
|
+
function shouldAnonymize(options) {
|
|
3141
|
+
if (options?.anonymizeJudge === false) return false;
|
|
3142
|
+
if (options?.noAnonymizeJudge === true) return false;
|
|
3143
|
+
return true;
|
|
3144
|
+
}
|
|
3145
|
+
function createAnonymizedVariantCopies(variants, variantToLabel) {
|
|
3146
|
+
return variants.map((variant) => {
|
|
3147
|
+
const label = variantToLabel.get(variant.variant);
|
|
3148
|
+
if (!label) return variant;
|
|
3149
|
+
return { ...variant, variant: label };
|
|
3150
|
+
});
|
|
3151
|
+
}
|
|
3152
|
+
function deAnonymizeRationale(text, idMapping, labelMapping) {
|
|
3153
|
+
let result = text;
|
|
3154
|
+
const allMappings = [
|
|
3155
|
+
...Object.entries(idMapping),
|
|
3156
|
+
...labelMapping ? Object.entries(labelMapping) : []
|
|
3157
|
+
];
|
|
3158
|
+
allMappings.sort((a, b) => b[0].length - a[0].length);
|
|
3159
|
+
for (const [anonymous, real] of allMappings) {
|
|
3160
|
+
const escaped = anonymous.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
3161
|
+
result = result.replace(new RegExp(`\\b${escaped}\\b`, "g"), real);
|
|
3162
|
+
}
|
|
3163
|
+
return result;
|
|
3164
|
+
}
|
|
3165
|
+
|
|
3166
|
+
// src/core/judge/duplicate-detector.ts
|
|
3167
|
+
var IMPORT_PATTERN = /^(?:import\b|export\s+(?:\*|\{[^}]*\})\s+from\b|const\s+[\w$]+\s*=\s*require\s*\(|require\s*\()/i;
|
|
3168
|
+
var LICENSE_HEADER_PATTERN = /^(?:\/\/|\/\*|\*|#)\s*(?:copyright|license|licensed under|spdx[- ]license[- ]identifier)/i;
|
|
3169
|
+
var TEST_SCAFFOLD_PATTERN = /^(?:describe|it|test|beforeEach|afterEach|beforeAll|afterAll)\s*\(/i;
|
|
3170
|
+
var TEST_HARNESS_PATTERN = /^(?:vi|jest)\.(?:mock|clearAllMocks|resetAllMocks|restoreAllMocks|useFakeTimers|useRealTimers)\s*\(/i;
|
|
3171
|
+
var STRUCTURAL_LINE_PATTERN = /^[()[\]{};,\s]+$/;
|
|
3172
|
+
var MIN_SUBSTANTIVE_LINES_FLOOR = 3;
|
|
3173
|
+
var MIN_SUBSTANTIVE_LINE_RATIO = 0.4;
|
|
3174
|
+
function fnv1aHash(str) {
|
|
3175
|
+
const FNV_OFFSET_BASIS = 2166136261;
|
|
3176
|
+
const FNV_PRIME = 16777619;
|
|
3177
|
+
let hash = FNV_OFFSET_BASIS;
|
|
3178
|
+
for (let i = 0; i < str.length; i++) {
|
|
3179
|
+
hash ^= str.charCodeAt(i);
|
|
3180
|
+
hash = Math.imul(hash, FNV_PRIME);
|
|
3181
|
+
}
|
|
3182
|
+
return hash >>> 0;
|
|
3183
|
+
}
|
|
3184
|
+
function normalizeLine(line) {
|
|
3185
|
+
const trimmed = line.trimEnd();
|
|
3186
|
+
const leadingMatch = trimmed.match(/^[\s]+/);
|
|
3187
|
+
if (!leadingMatch) return trimmed;
|
|
3188
|
+
const leadingSpaces = leadingMatch[0].replace(/\t/g, " ");
|
|
3189
|
+
return leadingSpaces + trimmed.trimStart();
|
|
3190
|
+
}
|
|
3191
|
+
function isBoilerplateLine(line) {
|
|
3192
|
+
const trimmed = line.trim();
|
|
3193
|
+
if (!trimmed) return true;
|
|
3194
|
+
if (IMPORT_PATTERN.test(trimmed)) return true;
|
|
3195
|
+
if (LICENSE_HEADER_PATTERN.test(trimmed)) return true;
|
|
3196
|
+
if (TEST_SCAFFOLD_PATTERN.test(trimmed)) return true;
|
|
3197
|
+
if (TEST_HARNESS_PATTERN.test(trimmed)) return true;
|
|
3198
|
+
if (STRUCTURAL_LINE_PATTERN.test(trimmed)) return true;
|
|
3199
|
+
return false;
|
|
3200
|
+
}
|
|
3201
|
+
function detectDuplicateContent(content, options = {}) {
|
|
3202
|
+
const minBlockLines = options.minBlockLines ?? 10;
|
|
3203
|
+
const threshold = options.threshold ?? 0.1;
|
|
3204
|
+
const minSubstantiveLines = Math.max(
|
|
3205
|
+
MIN_SUBSTANTIVE_LINES_FLOOR,
|
|
3206
|
+
Math.ceil(minBlockLines * MIN_SUBSTANTIVE_LINE_RATIO)
|
|
3207
|
+
);
|
|
3208
|
+
const lines = content.split("\n");
|
|
3209
|
+
if (lines.length < minBlockLines) {
|
|
3210
|
+
return {
|
|
3211
|
+
hasDuplicates: false,
|
|
3212
|
+
duplicationRatio: 0,
|
|
3213
|
+
duplicateBlockCount: 0,
|
|
3214
|
+
totalLines: lines.length,
|
|
3215
|
+
duplicatedLines: 0
|
|
3216
|
+
};
|
|
3217
|
+
}
|
|
3218
|
+
const normalizedLines = lines.map(normalizeLine);
|
|
3219
|
+
const blockMap = /* @__PURE__ */ new Map();
|
|
3220
|
+
for (let i = 0; i <= normalizedLines.length - minBlockLines; i++) {
|
|
3221
|
+
const blockLines = normalizedLines.slice(i, i + minBlockLines);
|
|
3222
|
+
const substantiveLineCount = blockLines.reduce(
|
|
3223
|
+
(count, line) => count + (isBoilerplateLine(line) ? 0 : 1),
|
|
3224
|
+
0
|
|
3225
|
+
);
|
|
3226
|
+
if (substantiveLineCount < minSubstantiveLines) {
|
|
3227
|
+
continue;
|
|
3228
|
+
}
|
|
3229
|
+
const blockContent = blockLines.join("\n");
|
|
3230
|
+
const hash = fnv1aHash(blockContent);
|
|
3231
|
+
if (!blockMap.has(hash)) {
|
|
3232
|
+
blockMap.set(hash, []);
|
|
3233
|
+
}
|
|
3234
|
+
blockMap.get(hash)?.push({ startLine: i, block: blockContent });
|
|
3235
|
+
}
|
|
3236
|
+
const duplicatePairs = [];
|
|
3237
|
+
for (const entries of blockMap.values()) {
|
|
3238
|
+
if (entries.length < 2) continue;
|
|
3239
|
+
const contentGroups = /* @__PURE__ */ new Map();
|
|
3240
|
+
for (const entry of entries) {
|
|
3241
|
+
if (!contentGroups.has(entry.block)) {
|
|
3242
|
+
contentGroups.set(entry.block, []);
|
|
3243
|
+
}
|
|
3244
|
+
contentGroups.get(entry.block)?.push(entry.startLine);
|
|
3245
|
+
}
|
|
3246
|
+
for (const [, positions] of contentGroups) {
|
|
3247
|
+
if (positions.length < 2) continue;
|
|
3248
|
+
const sortedPositions = [...positions].sort((a, b) => a - b);
|
|
3249
|
+
for (let i = 0; i < sortedPositions.length - 1; i++) {
|
|
3250
|
+
for (let j = i + 1; j < sortedPositions.length; j++) {
|
|
3251
|
+
const leftStart = sortedPositions[i];
|
|
3252
|
+
const rightStart = sortedPositions[j];
|
|
3253
|
+
if (rightStart < leftStart + minBlockLines) {
|
|
3254
|
+
continue;
|
|
3255
|
+
}
|
|
3256
|
+
let matchLength = minBlockLines;
|
|
3257
|
+
while (leftStart + matchLength < normalizedLines.length && rightStart + matchLength < normalizedLines.length && normalizedLines[leftStart + matchLength] === normalizedLines[rightStart + matchLength]) {
|
|
3258
|
+
matchLength++;
|
|
3259
|
+
}
|
|
3260
|
+
duplicatePairs.push({
|
|
3261
|
+
left: {
|
|
3262
|
+
startLine: leftStart,
|
|
3263
|
+
endLine: leftStart + matchLength - 1
|
|
3264
|
+
},
|
|
3265
|
+
right: {
|
|
3266
|
+
startLine: rightStart,
|
|
3267
|
+
endLine: rightStart + matchLength - 1
|
|
3268
|
+
}
|
|
3269
|
+
});
|
|
3270
|
+
}
|
|
3271
|
+
}
|
|
3272
|
+
}
|
|
3273
|
+
}
|
|
3274
|
+
duplicatePairs.sort((a, b) => {
|
|
3275
|
+
const aLength = a.left.endLine - a.left.startLine;
|
|
3276
|
+
const bLength = b.left.endLine - b.left.startLine;
|
|
3277
|
+
return bLength - aLength;
|
|
3278
|
+
});
|
|
3279
|
+
const selectedPairs = [];
|
|
3280
|
+
for (const pair of duplicatePairs) {
|
|
3281
|
+
const covered = selectedPairs.some(
|
|
3282
|
+
(selected) => pair.left.startLine >= selected.left.startLine && pair.left.endLine <= selected.left.endLine && pair.right.startLine >= selected.right.startLine && pair.right.endLine <= selected.right.endLine
|
|
3283
|
+
);
|
|
3284
|
+
if (!covered) {
|
|
3285
|
+
selectedPairs.push(pair);
|
|
3286
|
+
}
|
|
3287
|
+
}
|
|
3288
|
+
const duplicatedLineSet = /* @__PURE__ */ new Set();
|
|
3289
|
+
for (const pair of selectedPairs) {
|
|
3290
|
+
for (let line = pair.left.startLine; line <= pair.left.endLine; line++) {
|
|
3291
|
+
duplicatedLineSet.add(line);
|
|
3292
|
+
}
|
|
3293
|
+
for (let line = pair.right.startLine; line <= pair.right.endLine; line++) {
|
|
3294
|
+
duplicatedLineSet.add(line);
|
|
3295
|
+
}
|
|
3296
|
+
}
|
|
3297
|
+
const duplicatedLines = duplicatedLineSet.size;
|
|
3298
|
+
const totalLines = lines.length;
|
|
3299
|
+
const duplicationRatio = totalLines > 0 ? duplicatedLines / totalLines : 0;
|
|
3300
|
+
const duplicateBlockCount = selectedPairs.length;
|
|
3301
|
+
return {
|
|
3302
|
+
hasDuplicates: duplicationRatio >= threshold,
|
|
3303
|
+
duplicationRatio,
|
|
3304
|
+
duplicateBlockCount,
|
|
3305
|
+
totalLines,
|
|
3306
|
+
duplicatedLines
|
|
3307
|
+
};
|
|
3308
|
+
}
|
|
3309
|
+
|
|
3310
|
+
// src/core/judge/judge-failure-artifacts.ts
|
|
3311
|
+
import { mkdirSync, writeFileSync } from "fs";
|
|
3312
|
+
import path2 from "path";
|
|
3313
|
+
var DEFAULT_ARTIFACT_BASE_DIR = path2.join(".poetic", "artifacts");
|
|
3314
|
+
function safeLabPathComponent(value) {
|
|
3315
|
+
const trimmed = value?.trim();
|
|
3316
|
+
return trimmed && /^[A-Za-z0-9_-]+$/.test(trimmed) ? trimmed : void 0;
|
|
3317
|
+
}
|
|
3318
|
+
function defaultArtifactBaseDir() {
|
|
3319
|
+
const poeticDir = path2.join(process.cwd(), ".poetic");
|
|
3320
|
+
if (readEnv(POETIC_LAB_MODE).value !== true) {
|
|
3321
|
+
return path2.join(process.cwd(), DEFAULT_ARTIFACT_BASE_DIR);
|
|
3322
|
+
}
|
|
3323
|
+
const labId = safeLabPathComponent(readEnv(POETIC_LAB_ID).value);
|
|
3324
|
+
const runId = safeLabPathComponent(readEnv(POETIC_LAB_RUN_ID).value);
|
|
3325
|
+
if (!labId || !runId) {
|
|
3326
|
+
return path2.join(process.cwd(), DEFAULT_ARTIFACT_BASE_DIR);
|
|
3327
|
+
}
|
|
3328
|
+
return path2.join(poeticDir, "labs", labId, "runs", runId, "artifacts");
|
|
3329
|
+
}
|
|
3330
|
+
function persistJudgeFailureArtifact(args) {
|
|
3331
|
+
try {
|
|
3332
|
+
const safeTaskId = (args.taskId || "unknown").replace(/[^A-Za-z0-9._-]/g, "_");
|
|
3333
|
+
const timestamp = (/* @__PURE__ */ new Date()).toISOString().replace(/[:.]/g, "-");
|
|
3334
|
+
const artifactBase = args.artifactBaseDir ?? defaultArtifactBaseDir();
|
|
3335
|
+
const dir = path2.join(artifactBase, safeTaskId);
|
|
3336
|
+
mkdirSync(dir, { recursive: true });
|
|
3337
|
+
const filename = `judge-failure-${args.label}-${timestamp}.txt`;
|
|
3338
|
+
writeFileSync(path2.join(dir, filename), sanitizeString(args.response), "utf-8");
|
|
3339
|
+
} catch {
|
|
3340
|
+
}
|
|
3341
|
+
}
|
|
3342
|
+
|
|
3343
|
+
// src/core/judge/judge-model-normalizer.ts
|
|
3344
|
+
function resolveClaudeModelRole(role, fallbackRole) {
|
|
3345
|
+
try {
|
|
3346
|
+
return ProviderRegistry.getModelForRole("claude", role) ?? (fallbackRole ? ProviderRegistry.getModelForRole("claude", fallbackRole) : void 0) ?? ProviderRegistry.getDefaultModel("claude") ?? "sonnet";
|
|
3347
|
+
} catch {
|
|
3348
|
+
switch (role) {
|
|
3349
|
+
case "quality":
|
|
3350
|
+
return "opus";
|
|
3351
|
+
case "fast":
|
|
3352
|
+
return "haiku";
|
|
3353
|
+
default:
|
|
3354
|
+
return fallbackRole ?? "sonnet";
|
|
3355
|
+
}
|
|
3356
|
+
}
|
|
3357
|
+
}
|
|
3358
|
+
var CLAUDE_SONNET_DATED = resolveClaudeModelRole("judge_primary");
|
|
3359
|
+
var CLAUDE_OPUS_DATED = resolveClaudeModelRole("quality");
|
|
3360
|
+
var CLAUDE_HAIKU_DATED = resolveClaudeModelRole("fast");
|
|
3361
|
+
function getDefaultJudgeModel() {
|
|
3362
|
+
return resolveClaudeModelRole("judge_primary");
|
|
3363
|
+
}
|
|
3364
|
+
function getDefaultMultiLensJudgeModel() {
|
|
3365
|
+
return resolveClaudeModelRole("quality");
|
|
3366
|
+
}
|
|
3367
|
+
function normalizeJudgeModel(model, env = process.env) {
|
|
3368
|
+
const envJudgeModel = env.POETIC_JUDGE_MODEL?.trim();
|
|
3369
|
+
const envJudgeProvider = env.POETIC_JUDGE_PROVIDER?.trim();
|
|
3370
|
+
const resolved = resolveJudgeConfig(
|
|
3371
|
+
{
|
|
3372
|
+
judgeModel: model?.trim() || envJudgeModel,
|
|
3373
|
+
judgeProvider: envJudgeProvider
|
|
3374
|
+
},
|
|
3375
|
+
process.cwd()
|
|
3376
|
+
);
|
|
3377
|
+
const resolvedModel = String(resolved.model.value ?? "").trim();
|
|
3378
|
+
if (resolvedModel.length > 0 && resolvedModel !== "auto") {
|
|
3379
|
+
return {
|
|
3380
|
+
model: resolvedModel,
|
|
3381
|
+
provider: resolved.provider.value,
|
|
3382
|
+
warnings: []
|
|
3383
|
+
};
|
|
3384
|
+
}
|
|
3385
|
+
return { model: getDefaultJudgeModel(), provider: resolved.provider.value, warnings: [] };
|
|
3386
|
+
}
|
|
3387
|
+
|
|
3388
|
+
// src/core/judge/variant-result-coherence.ts
|
|
3389
|
+
import { existsSync } from "fs";
|
|
3390
|
+
function checkVariantResultCoherence(variantId, options) {
|
|
3391
|
+
const issues = [];
|
|
3392
|
+
let checksPerformed = 0;
|
|
3393
|
+
let checksPassed = 0;
|
|
3394
|
+
checksPerformed++;
|
|
3395
|
+
const parsed = parseVariantIdentifier(variantId || "");
|
|
3396
|
+
if (!parsed.provider) {
|
|
3397
|
+
issues.push({
|
|
3398
|
+
check: "variant_id_not_canonical",
|
|
3399
|
+
detail: `Variant ID "${variantId}" is not canonical (expected provider:model:strategy:mode format)`,
|
|
3400
|
+
severity: "warning"
|
|
3401
|
+
});
|
|
3402
|
+
} else {
|
|
3403
|
+
checksPassed++;
|
|
3404
|
+
}
|
|
3405
|
+
if (options?.checkWorktreePath && options.retainedWorktreePath) {
|
|
3406
|
+
checksPerformed++;
|
|
3407
|
+
if (existsSync(options.retainedWorktreePath)) {
|
|
3408
|
+
checksPassed++;
|
|
3409
|
+
} else {
|
|
3410
|
+
issues.push({
|
|
3411
|
+
check: "worktree_path_missing",
|
|
3412
|
+
detail: `Retained worktree path "${options.retainedWorktreePath}" does not exist on disk`,
|
|
3413
|
+
severity: "warning"
|
|
3414
|
+
});
|
|
3415
|
+
}
|
|
3416
|
+
}
|
|
3417
|
+
return {
|
|
3418
|
+
coherent: issues.length === 0,
|
|
3419
|
+
issues,
|
|
3420
|
+
checksPerformed,
|
|
3421
|
+
checksPassed
|
|
3422
|
+
};
|
|
3423
|
+
}
|
|
3424
|
+
|
|
3425
|
+
// src/core/judge/pre-score-gate-context.ts
|
|
3426
|
+
var SKIPPED_VERIFICATION_COVERAGE = /* @__PURE__ */ new Set([
|
|
3427
|
+
"skipped_not_relevant",
|
|
3428
|
+
"skipped_no_command",
|
|
3429
|
+
"skipped_tool_unavailable"
|
|
3430
|
+
]);
|
|
3431
|
+
function isSkippedVerificationCoverage(status) {
|
|
3432
|
+
return status !== void 0 && (SKIPPED_VERIFICATION_COVERAGE.has(status) || status === "not_run");
|
|
3433
|
+
}
|
|
3434
|
+
function shouldSuppressUnverifiedFullTestScope(tg) {
|
|
3435
|
+
if (tg.scope !== "full") return false;
|
|
3436
|
+
const filesTested = Array.isArray(tg.filesTested) ? tg.filesTested.filter((file) => typeof file === "string" && file.trim().length > 0) : [];
|
|
3437
|
+
if (filesTested.length > 0) return false;
|
|
3438
|
+
const command = typeof tg.command === "string" ? tg.command.trim() : "";
|
|
3439
|
+
if (!command) return false;
|
|
3440
|
+
return describeVerificationCommand(command).scope !== "full";
|
|
3441
|
+
}
|
|
3442
|
+
function resolveBuildGatePromptSignal(gate, gateSource) {
|
|
3443
|
+
if (gateSource === "unavailable" || gateSource === "execution_provenance") {
|
|
3444
|
+
return "unavailable";
|
|
3445
|
+
}
|
|
3446
|
+
const status = gate.verificationCoverage?.build.status ?? gate.buildGate.verificationCoverage?.status;
|
|
3447
|
+
if (isSkippedVerificationCoverage(status)) return "skipped";
|
|
3448
|
+
return gate.buildGate.signal;
|
|
3449
|
+
}
|
|
3450
|
+
function resolveLintGatePromptStatus(gate, gateSource) {
|
|
3451
|
+
if (gateSource === "unavailable" || gateSource === "execution_provenance") {
|
|
3452
|
+
return "unavailable";
|
|
3453
|
+
}
|
|
3454
|
+
const lintCoverageStatus = gate.verificationCoverage?.lint.status ?? gate.lintGate?.verificationCoverage?.status;
|
|
3455
|
+
if (isSkippedVerificationCoverage(lintCoverageStatus)) return "skipped";
|
|
3456
|
+
if (lintCoverageStatus === "unverified") return "unknown";
|
|
3457
|
+
if (gate.lintGate) return gate.lintGate.passed ? "pass" : "fail";
|
|
3458
|
+
return "not_run";
|
|
3459
|
+
}
|
|
3460
|
+
function resolveTestGatePromptSource(testGate, gateSource) {
|
|
3461
|
+
if (testGate?.evidenceSource === "reused_provenance") {
|
|
3462
|
+
return "execution_provenance";
|
|
3463
|
+
}
|
|
3464
|
+
return gateSource;
|
|
3465
|
+
}
|
|
3466
|
+
function isLexicalRequirementCoverageOnly(gateResults) {
|
|
3467
|
+
const { requirementGate } = gateResults;
|
|
3468
|
+
const details = Array.isArray(requirementGate.details) ? requirementGate.details : [];
|
|
3469
|
+
if (requirementGate.totalCount === 0 || details.length === 0 || details.length !== requirementGate.totalCount) {
|
|
3470
|
+
return false;
|
|
3471
|
+
}
|
|
3472
|
+
return details.every((detail) => detail.verificationDepth === "lexical");
|
|
3473
|
+
}
|
|
3474
|
+
function normalizeUnmeasuredLiveTestGate(gateResults) {
|
|
3475
|
+
const testGate = gateResults.testGate;
|
|
3476
|
+
if (testGate?.testStatus !== "passed" || testGate.metricsAvailable !== false) {
|
|
3477
|
+
return gateResults;
|
|
3478
|
+
}
|
|
3479
|
+
return {
|
|
3480
|
+
...gateResults,
|
|
3481
|
+
testGate: {
|
|
3482
|
+
...testGate,
|
|
3483
|
+
testStatus: "not_run",
|
|
3484
|
+
availability: "unavailable",
|
|
3485
|
+
skipReason: "no-supported-test-runner",
|
|
3486
|
+
testsRun: 0,
|
|
3487
|
+
testsPassed: 0,
|
|
3488
|
+
testsFailed: 0,
|
|
3489
|
+
exitCode: void 0
|
|
3490
|
+
},
|
|
3491
|
+
...gateResults.verificationCoverage ? {
|
|
3492
|
+
verificationCoverage: {
|
|
3493
|
+
...gateResults.verificationCoverage,
|
|
3494
|
+
test: {
|
|
3495
|
+
status: "skipped_no_command",
|
|
3496
|
+
command: testGate.command,
|
|
3497
|
+
reason: "Verification unavailable: the test command emitted no measurable test results."
|
|
3498
|
+
}
|
|
3499
|
+
}
|
|
3500
|
+
} : {}
|
|
3501
|
+
};
|
|
3502
|
+
}
|
|
3503
|
+
function renderRequirementGateSummary(gateResults, gateSource) {
|
|
3504
|
+
const source = gateSource === "live_gate" ? " [source=live_gate]" : "";
|
|
3505
|
+
const { requirementGate } = gateResults;
|
|
3506
|
+
const count = `(${requirementGate.satisfiedCount}/${requirementGate.totalCount})`;
|
|
3507
|
+
if (requirementGate.totalCount === 0) {
|
|
3508
|
+
return [
|
|
3509
|
+
`requirementGate: not_evaluated ${count}${source} [reason=no_requirements_parsed_from_prompt]`
|
|
3510
|
+
];
|
|
3511
|
+
}
|
|
3512
|
+
if (isLexicalRequirementCoverageOnly(gateResults)) {
|
|
3513
|
+
const status = requirementGate.satisfied ? "satisfied" : "partial";
|
|
3514
|
+
const lines = [
|
|
3515
|
+
`requirementGate: heuristic-${status} ${count}${source} [depth=lexical, heuristic=true]`
|
|
3516
|
+
];
|
|
3517
|
+
if (!requirementGate.satisfied) {
|
|
3518
|
+
lines.push(
|
|
3519
|
+
"requirementCoverageHeuristicNote: lexical keyword misses are not definitive acceptance failures; for behavioral requirements, prefer executed outcome evidence and directly named public API/test contracts."
|
|
3520
|
+
);
|
|
3521
|
+
}
|
|
3522
|
+
return lines;
|
|
3523
|
+
}
|
|
3524
|
+
return [
|
|
3525
|
+
`requirementGate: ${requirementGate.satisfied ? "satisfied" : "failed"} ${count}${source}`
|
|
3526
|
+
];
|
|
3527
|
+
}
|
|
3528
|
+
async function computeDeterministicPreScoreGateContext(variants, taskPrompt, options, loggerTag = "Judge", taskTypeResolution) {
|
|
3529
|
+
const gateResultsByVariant = /* @__PURE__ */ new Map();
|
|
3530
|
+
const structuralRisksByVariant = /* @__PURE__ */ new Map();
|
|
3531
|
+
const gateSourceByVariant = /* @__PURE__ */ new Map();
|
|
3532
|
+
const spec = options.evaluationContext?.promptSpec ?? parsePromptSpec(taskPrompt);
|
|
3533
|
+
const taskTypeBlocksCodeSignals = taskTypeResolution !== void 0 && (!taskTypeSignalsEnabled(taskTypeResolution) || taskTypeResolution.taskType !== "code");
|
|
3534
|
+
const likelyTechnical = !spec.isAnalysisIntent && (spec.fileConstraint.targets.length > 0 || spec.isTestIntent || spec.testExecutionRequested || /\b(code|typescript|javascript|python|test|build|lint|refactor|implement|implementation|function|class|module|bug|fix|endpoint|schema|migration)\b/i.test(
|
|
3535
|
+
taskPrompt
|
|
3536
|
+
));
|
|
3537
|
+
if (!likelyTechnical || variants.length === 0 || taskTypeBlocksCodeSignals) {
|
|
3538
|
+
return {
|
|
3539
|
+
gateResultsByVariant,
|
|
3540
|
+
structuralRisksByVariant,
|
|
3541
|
+
gateSourceByVariant,
|
|
3542
|
+
isCodeTask: false
|
|
3543
|
+
};
|
|
3544
|
+
}
|
|
3545
|
+
const requirements = spec.isAnalysisIntent ? [] : spec.requirements;
|
|
3546
|
+
const shouldRunBuildGate = requirements.length > 0 && variants.length > 1;
|
|
3547
|
+
const repoPath = options.repoPath || process.cwd();
|
|
3548
|
+
const baselineTypecheck = shouldRunBuildGate ? await (await import("./build-gate-YFASRB5Z.js")).collectBaselineTypecheck(repoPath) : null;
|
|
3549
|
+
const baselineHealthy = baselineTypecheck?.ok ?? false;
|
|
3550
|
+
const hasExecutionTestProvenance = variants.some(
|
|
3551
|
+
(variant) => hasTestExecutionProvenance(variant)
|
|
3552
|
+
);
|
|
3553
|
+
const shouldRunTestGate = Boolean(
|
|
3554
|
+
hasExecutionTestProvenance || spec.isTestIntent || spec.testExecutionRequested
|
|
3555
|
+
);
|
|
3556
|
+
const testGateMode = spec.testExecutionRequested ? "full" : "affected";
|
|
3557
|
+
const enrichedVariantsByName = /* @__PURE__ */ new Map();
|
|
3558
|
+
await Promise.all(
|
|
3559
|
+
variants.map(async (variant) => {
|
|
3560
|
+
const diffText = extractDiffText(variant);
|
|
3561
|
+
const changedFiles = extractChangedFilesMeta(diffText).map((meta) => meta.path);
|
|
3562
|
+
structuralRisksByVariant.set(variant.variant, scanDiffForStructuralRisks(diffText));
|
|
3563
|
+
const variantPath = options.variantPaths?.[variant.variant] || variant.retainedWorktreePath || null;
|
|
3564
|
+
try {
|
|
3565
|
+
let gateResults;
|
|
3566
|
+
let gateSource;
|
|
3567
|
+
if (!variantPath) {
|
|
3568
|
+
gateSource = hasTestExecutionProvenance(variant) ? "execution_provenance" : "unavailable";
|
|
3569
|
+
gateResults = buildDegradedGateResults(variant, baselineHealthy, gateSource);
|
|
3570
|
+
} else {
|
|
3571
|
+
gateSource = "live_gate";
|
|
3572
|
+
gateResults = normalizeUnmeasuredLiveTestGate(
|
|
3573
|
+
await runPreScoreGates(variantPath, changedFiles, requirements, baselineHealthy, {
|
|
3574
|
+
runBuildGate: shouldRunBuildGate,
|
|
3575
|
+
runLintGate: options.competitionMode === true && changedFiles.some((file) => categorizeFile(file) === "source"),
|
|
3576
|
+
variant,
|
|
3577
|
+
runTestGate: shouldRunTestGate,
|
|
3578
|
+
testGateMode,
|
|
3579
|
+
abortSignal: options.abortSignal,
|
|
3580
|
+
baselineTypecheck: baselineTypecheck ? {
|
|
3581
|
+
errorKeys: baselineTypecheck.errorKeys,
|
|
3582
|
+
typecheckExecutable: baselineTypecheck.typecheckExecutable
|
|
3583
|
+
} : void 0
|
|
3584
|
+
})
|
|
3585
|
+
);
|
|
3586
|
+
}
|
|
3587
|
+
gateResultsByVariant.set(variant.variant, gateResults);
|
|
3588
|
+
gateSourceByVariant.set(variant.variant, gateSource);
|
|
3589
|
+
const telemetryTaskId = options.taskId?.trim();
|
|
3590
|
+
if (telemetryTaskId) {
|
|
3591
|
+
options.telemetry?.logValidationResult?.({
|
|
3592
|
+
taskId: telemetryTaskId,
|
|
3593
|
+
variantId: variant.variant,
|
|
3594
|
+
gateResults
|
|
3595
|
+
});
|
|
3596
|
+
}
|
|
3597
|
+
const variantCopy = {
|
|
3598
|
+
...variant,
|
|
3599
|
+
empiricalSignals: variant.empiricalSignals ? { ...variant.empiricalSignals } : void 0
|
|
3600
|
+
};
|
|
3601
|
+
applyLiveTestGateProvenance(variantCopy, gateResults.testGate);
|
|
3602
|
+
if (variantCopy.empiricalSignals && gateResults.buildGate) {
|
|
3603
|
+
variantCopy.empiricalSignals.buildGate = {
|
|
3604
|
+
signal: gateResults.buildGate.signal,
|
|
3605
|
+
typeErrorCount: gateResults.buildGate.typeErrorCount,
|
|
3606
|
+
fatalErrorCount: gateResults.buildGate.fatalErrorCount,
|
|
3607
|
+
baselineHealthy: gateResults.buildGate.baselineHealthy,
|
|
3608
|
+
newErrorCount: gateResults.buildGate.newErrorCount,
|
|
3609
|
+
baselineComparisonAvailable: gateResults.buildGate.baselineComparisonAvailable,
|
|
3610
|
+
compilerCountsAvailable: gateResults.buildGate.compilerCountsAvailable
|
|
3611
|
+
};
|
|
3612
|
+
}
|
|
3613
|
+
if (variantCopy.empiricalSignals && gateResults.testGate) {
|
|
3614
|
+
variantCopy.empiricalSignals.testGate = {
|
|
3615
|
+
testStatus: gateResults.testGate.testStatus,
|
|
3616
|
+
testsRun: gateResults.testGate.testsRun,
|
|
3617
|
+
testsFailed: gateResults.testGate.testsFailed,
|
|
3618
|
+
testsPassed: gateResults.testGate.testsPassed,
|
|
3619
|
+
exitCode: gateResults.testGate.exitCode,
|
|
3620
|
+
signalConsistency: gateResults.testGate.signalConsistency
|
|
3621
|
+
};
|
|
3622
|
+
}
|
|
3623
|
+
{
|
|
3624
|
+
const findings = [];
|
|
3625
|
+
if (gateResults.buildGate.signal === "fail") {
|
|
3626
|
+
findings.push({
|
|
3627
|
+
kind: "build",
|
|
3628
|
+
command: gateResults.buildGate.command ?? "build",
|
|
3629
|
+
exitCode: 1,
|
|
3630
|
+
outputTail: gateResults.buildGate.stderrTail || gateResults.buildGate.outputTail,
|
|
3631
|
+
severity: "error"
|
|
3632
|
+
});
|
|
3633
|
+
}
|
|
3634
|
+
if (gateResults.testGate?.availability === "executed" && (gateResults.testGate.testStatus === "failed" || gateResults.testGate.testStatus === "error")) {
|
|
3635
|
+
findings.push({
|
|
3636
|
+
kind: "test",
|
|
3637
|
+
command: "test",
|
|
3638
|
+
exitCode: gateResults.testGate.exitCode ?? 1,
|
|
3639
|
+
outputTail: gateResults.testGate.outputTail,
|
|
3640
|
+
severity: "error"
|
|
3641
|
+
});
|
|
3642
|
+
}
|
|
3643
|
+
if (findings.length > 0) {
|
|
3644
|
+
const existing = variantCopy.verification?.findings ?? [];
|
|
3645
|
+
const merged = [...existing, ...findings];
|
|
3646
|
+
variantCopy.verification = {
|
|
3647
|
+
passed: !merged.some((finding) => finding.severity === "error"),
|
|
3648
|
+
findings: merged
|
|
3649
|
+
};
|
|
3650
|
+
}
|
|
3651
|
+
}
|
|
3652
|
+
if (variantCopy.empiricalSignals && JUDGE_FEATURE_FLAGS.VARIANT_DIFFERENTIATION_SIGNALS !== "off") {
|
|
3653
|
+
const coherence = checkVariantResultCoherence(variantCopy.variant, {
|
|
3654
|
+
retainedWorktreePath: variantCopy.retainedWorktreePath
|
|
3655
|
+
});
|
|
3656
|
+
variantCopy.empiricalSignals.coherenceCheck = {
|
|
3657
|
+
coherent: coherence.coherent,
|
|
3658
|
+
issues: coherence.issues.map((issue) => ({
|
|
3659
|
+
check: issue.check,
|
|
3660
|
+
detail: issue.detail
|
|
3661
|
+
}))
|
|
3662
|
+
};
|
|
3663
|
+
}
|
|
3664
|
+
enrichedVariantsByName.set(variantCopy.variant, variantCopy);
|
|
3665
|
+
} catch (error) {
|
|
3666
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
3667
|
+
console.warn(
|
|
3668
|
+
`[${loggerTag}] Pre-score gate evaluation failed for ${variant.variant}: ${message}`
|
|
3669
|
+
);
|
|
3670
|
+
gateResultsByVariant.delete(variant.variant);
|
|
3671
|
+
gateSourceByVariant.delete(variant.variant);
|
|
3672
|
+
}
|
|
3673
|
+
})
|
|
3674
|
+
);
|
|
3675
|
+
return {
|
|
3676
|
+
gateResultsByVariant,
|
|
3677
|
+
structuralRisksByVariant,
|
|
3678
|
+
gateSourceByVariant,
|
|
3679
|
+
isCodeTask: true,
|
|
3680
|
+
enrichedVariants: variants.map(
|
|
3681
|
+
(variant) => enrichedVariantsByName.get(variant.variant) ?? variant
|
|
3682
|
+
)
|
|
3683
|
+
};
|
|
3684
|
+
}
|
|
3685
|
+
function applyLiveTestGateProvenance(variant, testGate) {
|
|
3686
|
+
if (!testGate || testGate.testStatus === "not_run") return;
|
|
3687
|
+
const hasExistingExecutedTests = Boolean(
|
|
3688
|
+
variant.testsExecuted && variant.testsExecuted.scope !== "none"
|
|
3689
|
+
);
|
|
3690
|
+
const liveGateFailed = testGate.testStatus === "failed" || testGate.testStatus === "error" || testGate.testStatus === "timeout";
|
|
3691
|
+
const existingFailed = variant.testStatus === "failed" || variant.testStatus === "error" || variant.testStatus === "timeout";
|
|
3692
|
+
const liveGatePassed = testGate.testStatus === "passed";
|
|
3693
|
+
if (hasExistingExecutedTests && (variant.testStatus === "passed" && liveGateFailed || existingFailed && liveGatePassed)) {
|
|
3694
|
+
return;
|
|
3695
|
+
}
|
|
3696
|
+
variant.testStatus = testGate.testStatus;
|
|
3697
|
+
variant.testMetrics = {
|
|
3698
|
+
totalTests: testGate.testsRun,
|
|
3699
|
+
passedTests: testGate.testsPassed,
|
|
3700
|
+
failedTests: testGate.testsFailed,
|
|
3701
|
+
durationMs: testGate.duration,
|
|
3702
|
+
passRate: testGate.testsRun > 0 ? testGate.testsPassed / testGate.testsRun * 100 : void 0
|
|
3703
|
+
};
|
|
3704
|
+
variant.testsExecuted = {
|
|
3705
|
+
scope: testGate.scope,
|
|
3706
|
+
filesTested: testGate.filesTested,
|
|
3707
|
+
command: testGate.command,
|
|
3708
|
+
...testGate.commands && testGate.commands.length > 0 ? { commands: testGate.commands } : {},
|
|
3709
|
+
exitCode: testGate.exitCode,
|
|
3710
|
+
durationMs: testGate.duration,
|
|
3711
|
+
...testGate.failingTests && testGate.failingTests.length > 0 ? { failingTests: testGate.failingTests } : {},
|
|
3712
|
+
...testGate.outputTail ? { outputTail: testGate.outputTail } : {}
|
|
3713
|
+
};
|
|
3714
|
+
}
|
|
3715
|
+
function buildDegradedGateResults(variant, baselineHealthy, gateSource) {
|
|
3716
|
+
let testGate;
|
|
3717
|
+
const testsExecuted = variant.testsExecuted;
|
|
3718
|
+
let buildPassed = false;
|
|
3719
|
+
if (gateSource === "execution_provenance" && testsExecuted && testsExecuted.scope !== "none") {
|
|
3720
|
+
const scope = testsExecuted.scope === "targeted" ? "touched" : testsExecuted.scope;
|
|
3721
|
+
const exitCode = testsExecuted.exitCode;
|
|
3722
|
+
const isEnvCommandFailure = (code) => code === 126 || code === 127;
|
|
3723
|
+
const recordedStatus = normalizeRecordedTestStatus(variant.testStatus);
|
|
3724
|
+
const recordedFailure = recordedStatus === "failed" || recordedStatus === "error" || recordedStatus === "timeout";
|
|
3725
|
+
const inferredStatus = recordedStatus === "not_run" ? "not_run" : (
|
|
3726
|
+
// Fail closed against a clean recorded exit code in two ways, matching the
|
|
3727
|
+
// ordering the live reuse path in test-gate.ts already uses (a recorded
|
|
3728
|
+
// status wins; the exit code only fills a gap):
|
|
3729
|
+
// - a recorded FAILURE is not masked by `exitCode: 0`;
|
|
3730
|
+
// - a status outside the interpretable vocabulary is not resolved to
|
|
3731
|
+
// `passed` by `exitCode: 0` either. Failure inference from a non-zero
|
|
3732
|
+
// exit stays unchanged.
|
|
3733
|
+
recordedFailure ? recordedStatus : hasUnrecognizedRecordedTestStatus(variant.testStatus) && exitCode === 0 ? "not_run" : exitCode === 0 ? "passed" : exitCode === 124 ? "timeout" : isEnvCommandFailure(exitCode) ? "error" : exitCode === void 0 ? recordedStatus ?? "not_run" : "failed"
|
|
3734
|
+
);
|
|
3735
|
+
const failedTests = variant.testMetrics?.failedTests ?? 0;
|
|
3736
|
+
const totalTests = variant.testMetrics?.totalTests ?? 0;
|
|
3737
|
+
const hasExecutionEvidence = hasRecordedTestExecutionEvidence({
|
|
3738
|
+
command: testsExecuted.command,
|
|
3739
|
+
filesTested: testsExecuted.filesTested,
|
|
3740
|
+
totalTests,
|
|
3741
|
+
exitCode: testsExecuted.exitCode
|
|
3742
|
+
});
|
|
3743
|
+
const normalizedStatus = hasExecutionEvidence && !(inferredStatus === "passed" && totalTests === 0) ? inferredStatus : "not_run";
|
|
3744
|
+
const isContradiction = inferredStatus === "failed" && failedTests === 0 && totalTests > 0;
|
|
3745
|
+
testGate = {
|
|
3746
|
+
testStatus: normalizedStatus,
|
|
3747
|
+
availability: normalizedStatus === "not_run" ? "unavailable" : "executed",
|
|
3748
|
+
skipReason: normalizedStatus === "not_run" ? "no-execution-provenance" : void 0,
|
|
3749
|
+
testsRun: normalizedStatus === "not_run" ? 0 : totalTests,
|
|
3750
|
+
testsPassed: normalizedStatus === "not_run" ? 0 : variant.testMetrics?.passedTests ?? 0,
|
|
3751
|
+
testsFailed: normalizedStatus === "not_run" ? 0 : failedTests,
|
|
3752
|
+
// A recorded count is the only measurement evidence available on the
|
|
3753
|
+
// degraded (no-worktree) path; without it the zeros above are "unknown",
|
|
3754
|
+
// not "zero tests ran".
|
|
3755
|
+
...normalizedStatus === "not_run" ? { metricsAvailable: false } : { metricsAvailable: true },
|
|
3756
|
+
duration: testsExecuted.durationMs ?? variant.testMetrics?.durationMs ?? 0,
|
|
3757
|
+
filesTested: normalizedStatus === "not_run" ? [] : testsExecuted.filesTested ?? [],
|
|
3758
|
+
command: normalizedStatus === "not_run" ? "" : testsExecuted.command ?? "",
|
|
3759
|
+
scope: normalizedStatus === "not_run" ? "none" : scope,
|
|
3760
|
+
exitCode: normalizedStatus === "not_run" ? void 0 : exitCode,
|
|
3761
|
+
outputTail: testsExecuted.outputTail,
|
|
3762
|
+
failingTests: testsExecuted.failingTests ?? [],
|
|
3763
|
+
signalConsistency: isContradiction ? "inconsistent" : "consistent"
|
|
3764
|
+
};
|
|
3765
|
+
buildPassed = normalizedStatus === "passed";
|
|
3766
|
+
}
|
|
3767
|
+
const buildGate = {
|
|
3768
|
+
signal: "unknown",
|
|
3769
|
+
errors: [],
|
|
3770
|
+
baselineComparisonAvailable: false,
|
|
3771
|
+
baselineHealthy,
|
|
3772
|
+
fatalErrorCount: 0,
|
|
3773
|
+
typeErrorCount: 0,
|
|
3774
|
+
warningCount: 0,
|
|
3775
|
+
verificationCoverage: {
|
|
3776
|
+
status: "skipped_no_command",
|
|
3777
|
+
reason: "Variant worktree unavailable."
|
|
3778
|
+
}
|
|
3779
|
+
};
|
|
3780
|
+
return {
|
|
3781
|
+
// Requirement validation needs the variant worktree. Keep the typed
|
|
3782
|
+
// compatibility shape neutral; gateSource carries the unavailable state.
|
|
3783
|
+
requirementGate: {
|
|
3784
|
+
satisfied: true,
|
|
3785
|
+
satisfiedCount: 0,
|
|
3786
|
+
totalCount: 0,
|
|
3787
|
+
missingRequirements: [],
|
|
3788
|
+
details: []
|
|
3789
|
+
},
|
|
3790
|
+
buildGate,
|
|
3791
|
+
lintGate: void 0,
|
|
3792
|
+
testGate,
|
|
3793
|
+
verificationCoverage: {
|
|
3794
|
+
build: {
|
|
3795
|
+
status: "skipped_no_command",
|
|
3796
|
+
reason: "Variant worktree unavailable."
|
|
3797
|
+
},
|
|
3798
|
+
lint: {
|
|
3799
|
+
status: "skipped_no_command",
|
|
3800
|
+
reason: "Variant worktree unavailable."
|
|
3801
|
+
}
|
|
3802
|
+
},
|
|
3803
|
+
buildPassed,
|
|
3804
|
+
lintPassed: false,
|
|
3805
|
+
isEligible: true
|
|
3806
|
+
};
|
|
3807
|
+
}
|
|
3808
|
+
function clipSnippet2(text, maxChars = 500) {
|
|
3809
|
+
if (typeof text !== "string") return null;
|
|
3810
|
+
const trimmed = text.trim();
|
|
3811
|
+
if (trimmed.length === 0) return null;
|
|
3812
|
+
if (trimmed.length <= maxChars) return trimmed;
|
|
3813
|
+
return `${trimmed.slice(0, maxChars)}...`;
|
|
3814
|
+
}
|
|
3815
|
+
function buildDeterministicPreScoreGateSection(gateResults, structuralRisks, options) {
|
|
3816
|
+
if (!gateResults && !structuralRisks) return void 0;
|
|
3817
|
+
const gateSource = options?.gateSource;
|
|
3818
|
+
const noVariantPath = gateSource === "execution_provenance" || gateSource === "unavailable";
|
|
3819
|
+
const lines = [];
|
|
3820
|
+
if (gateResults) {
|
|
3821
|
+
if (noVariantPath) {
|
|
3822
|
+
lines.push("requirementGate: unavailable [source=unavailable, reason=no_variant_path]");
|
|
3823
|
+
} else {
|
|
3824
|
+
lines.push(...renderRequirementGateSummary(gateResults, gateSource));
|
|
3825
|
+
}
|
|
3826
|
+
if (gateSource === "live_gate") {
|
|
3827
|
+
const buildSignal = resolveBuildGatePromptSignal(gateResults, gateSource);
|
|
3828
|
+
const buildSkipped = buildSignal === "skipped";
|
|
3829
|
+
const compilerCountsMeasured = gateResults.buildGate.compilerCountsAvailable !== false;
|
|
3830
|
+
const buildDetails = !options?.includeCompilerCounts ? "" : compilerCountsMeasured ? ` (typeErrorCount: ${gateResults.buildGate.typeErrorCount}, fatalErrorCount: ${gateResults.buildGate.fatalErrorCount}${typeof gateResults.buildGate.newErrorCount === "number" ? `, newErrorCount: ${gateResults.buildGate.newErrorCount}` : ""})` : " (compilerCounts: unavailable, non-Node build output not parsed)";
|
|
3831
|
+
lines.push(`buildGate: ${buildSignal}${buildSkipped ? "" : buildDetails} [source=live_gate]`);
|
|
3832
|
+
} else {
|
|
3833
|
+
lines.push("buildGate: unavailable [source=unavailable, reason=no_variant_path]");
|
|
3834
|
+
}
|
|
3835
|
+
const tg = gateResults.testGate;
|
|
3836
|
+
const testSource = resolveTestGatePromptSource(tg, gateSource);
|
|
3837
|
+
if (!tg && gateSource === "unavailable") {
|
|
3838
|
+
lines.push("testGate: unavailable [source=unavailable, reason=no_verified_test_provenance]");
|
|
3839
|
+
} else {
|
|
3840
|
+
const suppressFullScope = tg ? shouldSuppressUnverifiedFullTestScope(tg) : false;
|
|
3841
|
+
const renderedScope = tg?.scope && tg.scope !== "none" && !suppressFullScope ? `, scope=${tg.scope}` : "";
|
|
3842
|
+
const renderedSource = testSource ? ` [source=${testSource}${renderedScope}]` : tg?.scope && tg.scope !== "none" && !suppressFullScope ? ` [scope=${tg.scope}]` : "";
|
|
3843
|
+
const command = typeof tg?.command === "string" && tg.command.trim().length > 0 ? escapeEvidenceForJudgePromptDisplay(tg.command.trim()) : void 0;
|
|
3844
|
+
const commandDetail = suppressFullScope && command ? `, command: ${command}` : "";
|
|
3845
|
+
lines.push(
|
|
3846
|
+
`testGate: ${tg?.testStatus ?? "not_run"}${tg ? ` (testsRun: ${tg.testsRun}, testsFailed: ${tg.testsFailed}, testsPassed: ${tg.testsPassed}${tg.exitCode !== void 0 ? `, exitCode: ${tg.exitCode}` : ""}${tg.metricsAvailable === false ? ", metricsAvailable: false" : ""}${commandDetail})${renderedSource}` : renderedSource}`
|
|
3847
|
+
);
|
|
3848
|
+
if (suppressFullScope && command) {
|
|
3849
|
+
lines.push(
|
|
3850
|
+
`WARNING: recorded test scope was "full" but the executed command is not classified as a full suite and selectedFiles were empty. Do not treat this as evidence that the variant's new tests ran.`
|
|
3851
|
+
);
|
|
3852
|
+
}
|
|
3853
|
+
}
|
|
3854
|
+
if (tg?.testStatus === "passed" && tg.metricsAvailable === false) {
|
|
3855
|
+
lines.push(
|
|
3856
|
+
'WARNING: the test command exited 0 but emitted no parseable test counts, so testsRun/testsPassed above are "unknown", not measured zeros. This is not evidence that tests executed; do not cite it as verified test coverage.'
|
|
3857
|
+
);
|
|
3858
|
+
}
|
|
3859
|
+
if (tg?.signalConsistency === "inconsistent") {
|
|
3860
|
+
lines.push(
|
|
3861
|
+
"WARNING: exitCode indicates failure but parser found 0 failing tests. Likely infrastructure/tooling error, not a clean test pass. Do not interpret parsed metrics as evidence of passing tests."
|
|
3862
|
+
);
|
|
3863
|
+
}
|
|
3864
|
+
const lintStatus = resolveLintGatePromptStatus(gateResults, gateSource);
|
|
3865
|
+
if (gateSource === "live_gate") {
|
|
3866
|
+
lines.push(`lintGate: ${lintStatus} [source=live_gate]`);
|
|
3867
|
+
} else {
|
|
3868
|
+
lines.push("lintGate: unavailable [source=unavailable, reason=no_variant_path]");
|
|
3869
|
+
}
|
|
3870
|
+
if (lintStatus === "fail") {
|
|
3871
|
+
lines.push(
|
|
3872
|
+
"(Lint primarily enforces style/formatting rules. Not a functional correctness signal unless a specific rule flags a defect.)"
|
|
3873
|
+
);
|
|
3874
|
+
}
|
|
3875
|
+
lines.push(`gateEligibilityObservation: ${gateResults.isEligible} [telemetry_only]`);
|
|
3876
|
+
if (gateResults.diagnosticReason) {
|
|
3877
|
+
lines.push(
|
|
3878
|
+
`gateEligibilityReasonObservation: ${gateResults.diagnosticReason} [telemetry_only]`
|
|
3879
|
+
);
|
|
3880
|
+
}
|
|
3881
|
+
}
|
|
3882
|
+
if (options?.isCodeTask) {
|
|
3883
|
+
if (structuralRisks) {
|
|
3884
|
+
if (structuralRisks.applicability === "not_applicable") {
|
|
3885
|
+
lines.push(
|
|
3886
|
+
"structuralRisks: not_applicable (heuristic targets JS/TS syntax; unsupported-language absence is not checked-clean) [source=heuristic, method=diff-pattern-matching, no AST]"
|
|
3887
|
+
);
|
|
3888
|
+
} else if (structuralRisks.applicability === "unknown" && structuralRisks.findings.length === 0) {
|
|
3889
|
+
lines.push(
|
|
3890
|
+
"structuralRisks: unknown (no classifiable source paths for structural-risk heuristics) [source=heuristic, method=diff-pattern-matching, no AST]"
|
|
3891
|
+
);
|
|
3892
|
+
} else {
|
|
3893
|
+
const applicabilityNote = structuralRisks.applicability === "mixed" ? " applicability=mixed (only JS/TS paths scanned)" : "";
|
|
3894
|
+
lines.push(
|
|
3895
|
+
`structuralRisks: critical=${structuralRisks.criticalCount}, major=${structuralRisks.majorCount}, minor=${structuralRisks.minorCount}${applicabilityNote} [source=heuristic, method=diff-pattern-matching, no AST. Corroborate with build/test before weighing this signal. False positives expected on large refactors.]`
|
|
3896
|
+
);
|
|
3897
|
+
if (structuralRisks.findings.length > 0) {
|
|
3898
|
+
lines.push("Structural Risk Signals:");
|
|
3899
|
+
for (const finding of structuralRisks.findings.slice(0, 3)) {
|
|
3900
|
+
const displayPath = finding.filePath !== void 0 ? escapePathForJudgePromptDisplay(finding.filePath) : void 0;
|
|
3901
|
+
const location = displayPath && finding.line ? `${displayPath}:${finding.line}` : displayPath ?? "unknown location";
|
|
3902
|
+
lines.push(
|
|
3903
|
+
` - ${finding.severity.toUpperCase()}: ${finding.summary} (${location}) [heuristic: diff-only brace-counting, no AST]`
|
|
3904
|
+
);
|
|
3905
|
+
if (finding.evidence) {
|
|
3906
|
+
lines.push(
|
|
3907
|
+
` evidence: ${escapeEvidenceForJudgePromptDisplay(finding.evidence.slice(0, 200))}`
|
|
3908
|
+
);
|
|
3909
|
+
}
|
|
3910
|
+
}
|
|
3911
|
+
}
|
|
3912
|
+
}
|
|
3913
|
+
}
|
|
3914
|
+
const testFailureSnippet = clipSnippet2(gateResults?.testGate?.outputTail, 500);
|
|
3915
|
+
if (gateResults && (testFailureSnippet || (gateResults.testGate?.failingTests?.length ?? 0) > 0) && (gateResults.testGate?.testStatus === "failed" || gateResults.testGate?.testStatus === "timeout" || gateResults.testGate?.testStatus === "error")) {
|
|
3916
|
+
const failingTests = gateResults.testGate?.failingTests;
|
|
3917
|
+
if (failingTests && failingTests.length > 0) {
|
|
3918
|
+
const maxFailedTestNames = 5;
|
|
3919
|
+
const failedTestPreview = failingTests.slice(0, maxFailedTestNames).map(escapeEvidenceForJudgePromptDisplay);
|
|
3920
|
+
const remainingFailedTests = failingTests.length - failedTestPreview.length;
|
|
3921
|
+
const suffix = remainingFailedTests > 0 ? ` (+${remainingFailedTests} more)` : "";
|
|
3922
|
+
lines.push(`failedTestNames: ${failedTestPreview.join(", ")}${suffix}`);
|
|
3923
|
+
}
|
|
3924
|
+
if (testFailureSnippet) {
|
|
3925
|
+
lines.push("testFailureDetails:");
|
|
3926
|
+
lines.push(
|
|
3927
|
+
...testFailureSnippet.split("\n").map((line) => ` ${escapeEvidenceForJudgePromptDisplay(line)}`)
|
|
3928
|
+
);
|
|
3929
|
+
}
|
|
3930
|
+
lines.push(
|
|
3931
|
+
"redResultClassification: required \u2014 classify this result as production_defect, stale_fixture, intentional_contract_change, or infrastructure_failure, citing the supplied failure evidence."
|
|
3932
|
+
);
|
|
3933
|
+
}
|
|
3934
|
+
if (gateResults) {
|
|
3935
|
+
const buildFailureSnippet = clipSnippet2(
|
|
3936
|
+
gateResults.buildGate.stderrTail || gateResults.buildGate.outputTail || gateResults.buildGate.stdoutTail,
|
|
3937
|
+
500
|
|
3938
|
+
);
|
|
3939
|
+
if (buildFailureSnippet && gateResults.buildGate.signal === "fail") {
|
|
3940
|
+
lines.push("buildFailureDetails:");
|
|
3941
|
+
lines.push(
|
|
3942
|
+
...buildFailureSnippet.split("\n").map((line) => ` ${escapeEvidenceForJudgePromptDisplay(line)}`)
|
|
3943
|
+
);
|
|
3944
|
+
}
|
|
3945
|
+
}
|
|
3946
|
+
}
|
|
3947
|
+
return lines.join("\n");
|
|
3948
|
+
}
|
|
3949
|
+
|
|
3950
|
+
// src/core/judge/ranking-validation.ts
|
|
3951
|
+
function normalizeRankingIds(rawRanking, variantsList) {
|
|
3952
|
+
const exactVariants = new Set(variantsList.map((variant) => variant.variant));
|
|
3953
|
+
return rawRanking.map((entry) => {
|
|
3954
|
+
const trimmed = entry.trim();
|
|
3955
|
+
if (exactVariants.has(trimmed)) return trimmed;
|
|
3956
|
+
const idMatch = trimmed.match(/^v(\d+)$/i);
|
|
3957
|
+
if (!idMatch) return trimmed;
|
|
3958
|
+
const idx = Number(idMatch[1]) - 1;
|
|
3959
|
+
if (!Number.isFinite(idx) || idx < 0 || idx >= variantsList.length) {
|
|
3960
|
+
return trimmed;
|
|
3961
|
+
}
|
|
3962
|
+
return variantsList[idx]?.variant ?? trimmed;
|
|
3963
|
+
});
|
|
3964
|
+
}
|
|
3965
|
+
function validateRanking(ranking, knownVariants, variantsList) {
|
|
3966
|
+
if (!Array.isArray(ranking)) {
|
|
3967
|
+
return { valid: false, reason: "ranking_not_array" };
|
|
3968
|
+
}
|
|
3969
|
+
if (ranking.length === 0) {
|
|
3970
|
+
return { valid: false, reason: "ranking_empty" };
|
|
3971
|
+
}
|
|
3972
|
+
if (!ranking.every((entry) => typeof entry === "string" && entry.trim().length > 0)) {
|
|
3973
|
+
return { valid: false, reason: "ranking_non_string" };
|
|
3974
|
+
}
|
|
3975
|
+
const raw = ranking.map((entry) => String(entry).trim());
|
|
3976
|
+
const normalized = variantsList ? normalizeRankingIds(raw, variantsList) : raw;
|
|
3977
|
+
const seen = /* @__PURE__ */ new Set();
|
|
3978
|
+
for (const entry of normalized) {
|
|
3979
|
+
if (entry.trim().length === 0) {
|
|
3980
|
+
return { valid: false, reason: "ranking_non_string" };
|
|
3981
|
+
}
|
|
3982
|
+
if (seen.has(entry)) {
|
|
3983
|
+
return { valid: false, reason: "ranking_duplicate_variant" };
|
|
3984
|
+
}
|
|
3985
|
+
seen.add(entry);
|
|
3986
|
+
if (!knownVariants.has(entry)) {
|
|
3987
|
+
return { valid: false, reason: "ranking_unknown_variant" };
|
|
3988
|
+
}
|
|
3989
|
+
}
|
|
3990
|
+
if (normalized.length > knownVariants.size) {
|
|
3991
|
+
return { valid: false, reason: "ranking_length_mismatch" };
|
|
3992
|
+
}
|
|
3993
|
+
const partial = normalized.length < knownVariants.size;
|
|
3994
|
+
return { valid: true, ranking: normalized, partial };
|
|
3995
|
+
}
|
|
3996
|
+
function checkRankingScoreConsistency(ranking, scoresByVariant) {
|
|
3997
|
+
for (let i = 0; i < ranking.length - 1; i++) {
|
|
3998
|
+
const current = ranking[i];
|
|
3999
|
+
const next = ranking[i + 1];
|
|
4000
|
+
const currentScore = scoresByVariant.get(current);
|
|
4001
|
+
const nextScore = scoresByVariant.get(next);
|
|
4002
|
+
if (currentScore === void 0 || nextScore === void 0) continue;
|
|
4003
|
+
if (nextScore > currentScore) {
|
|
4004
|
+
return {
|
|
4005
|
+
inconsistent: true,
|
|
4006
|
+
details: `rk[${i}]="${current}" (os=${currentScore}) < rk[${i + 1}]="${next}" (os=${nextScore})`
|
|
4007
|
+
};
|
|
4008
|
+
}
|
|
4009
|
+
}
|
|
4010
|
+
return { inconsistent: false };
|
|
4011
|
+
}
|
|
4012
|
+
|
|
4013
|
+
// src/core/judge/intent-utils.ts
|
|
4014
|
+
function isAnalysisIntent(variant) {
|
|
4015
|
+
return variant.executionIntent === "analysis" || variant.executionCategory === "analysis" || variant.completenessScore?.editMode === "analysis";
|
|
4016
|
+
}
|
|
4017
|
+
|
|
4018
|
+
// src/core/judge/static-planning-detection.ts
|
|
4019
|
+
function detectPlanningCategoryStatic(variant) {
|
|
4020
|
+
const filesChanged = variant.emptyChanges ? 0 : variant.gitMetrics?.filesChanged ?? variant.filesChanged ?? variant.worktreeSummary?.filesChanged ?? 0;
|
|
4021
|
+
const output = extractCleanStdout(variant);
|
|
4022
|
+
const planningKeywords = [
|
|
4023
|
+
"approach:",
|
|
4024
|
+
"strategy:",
|
|
4025
|
+
"plan:",
|
|
4026
|
+
"step-by-step",
|
|
4027
|
+
"step by step",
|
|
4028
|
+
"implementation steps",
|
|
4029
|
+
"steps:",
|
|
4030
|
+
"todo",
|
|
4031
|
+
"fixme",
|
|
4032
|
+
"consideration:",
|
|
4033
|
+
"recommendation:",
|
|
4034
|
+
"1.",
|
|
4035
|
+
"2.",
|
|
4036
|
+
"3.",
|
|
4037
|
+
"first,",
|
|
4038
|
+
"second,",
|
|
4039
|
+
"then,",
|
|
4040
|
+
"finally,",
|
|
4041
|
+
"would need to",
|
|
4042
|
+
"should be",
|
|
4043
|
+
"could implement"
|
|
4044
|
+
];
|
|
4045
|
+
const lowerOutput = output.toLowerCase();
|
|
4046
|
+
const keywordMatches = planningKeywords.filter((kw) => lowerOutput.includes(kw)).length;
|
|
4047
|
+
const hasStepHeadings = /\bstep\s+\d+\b/i.test(output);
|
|
4048
|
+
const bulletCount = (output.match(/^\s*[-*]\s+/gm) ?? []).length;
|
|
4049
|
+
const numberedLineCount = (output.match(/^\s*\d+[).]\s+/gm) ?? []).length;
|
|
4050
|
+
const numberedHeadingCount = (output.match(/^\s*#{1,6}\s*\d+[).:]\s+/gm) ?? []).length;
|
|
4051
|
+
const stepHeadingCount = (output.match(/^\s*#{1,6}\s*step\s+\d+\b/gim) ?? []).length;
|
|
4052
|
+
const hasStructuredList = bulletCount >= 3 || numberedLineCount >= 3 || numberedHeadingCount >= 2 || stepHeadingCount >= 2;
|
|
4053
|
+
const isLikelyPlanning = filesChanged === 0 && // Keep this conservative to avoid false positives:
|
|
4054
|
+
// - Require substantial output length (> 500 chars), not just a short note.
|
|
4055
|
+
// - Require several planning signals (3+ keyword hits) or a clearly structured list.
|
|
4056
|
+
output.length > 500 && (keywordMatches >= 3 || hasStepHeadings || hasStructuredList);
|
|
4057
|
+
return isLikelyPlanning ? "planning" : "implementation";
|
|
4058
|
+
}
|
|
4059
|
+
|
|
4060
|
+
// src/core/judge/output-inspection-gate.ts
|
|
4061
|
+
var MIN_STDOUT_LENGTH = 80;
|
|
4062
|
+
function inspectVariantOutput(variant, options = {}) {
|
|
4063
|
+
const reasons = [];
|
|
4064
|
+
const { editsExpected = true } = options;
|
|
4065
|
+
const cleanStdout = extractCleanStdout(variant);
|
|
4066
|
+
const filesChanged = variant.gitMetrics?.filesChanged ?? variant.filesChanged ?? 0;
|
|
4067
|
+
const hasWorktreeDiff = Boolean(variant.worktreeDiff && variant.worktreeDiff.length > 0);
|
|
4068
|
+
const diffText = hasWorktreeDiff ? extractDiffText(variant) : "";
|
|
4069
|
+
const evidence = {
|
|
4070
|
+
stdoutLength: cleanStdout.length,
|
|
4071
|
+
diffLength: diffText.length,
|
|
4072
|
+
filesChanged,
|
|
4073
|
+
hasWorktreeDiff
|
|
4074
|
+
};
|
|
4075
|
+
let passed = false;
|
|
4076
|
+
const stdoutOnlyPlanningForCodeTask = editsExpected && cleanStdout.length > 0 && diffText.length === 0 && filesChanged === 0 && detectPlanningCategoryStatic(variant) === "planning";
|
|
4077
|
+
if (cleanStdout.length >= MIN_STDOUT_LENGTH && !stdoutOnlyPlanningForCodeTask) {
|
|
4078
|
+
passed = true;
|
|
4079
|
+
reasons.push(`stdout >= ${MIN_STDOUT_LENGTH} chars (${cleanStdout.length})`);
|
|
4080
|
+
}
|
|
4081
|
+
if (cleanStdout.length > 0 && cleanStdout.length < MIN_STDOUT_LENGTH && !stdoutOnlyPlanningForCodeTask) {
|
|
4082
|
+
passed = true;
|
|
4083
|
+
reasons.push(`stdout > 0 chars (${cleanStdout.length})`);
|
|
4084
|
+
}
|
|
4085
|
+
if (stdoutOnlyPlanningForCodeTask) {
|
|
4086
|
+
reasons.push("stdout-only output classified as planning for code task");
|
|
4087
|
+
}
|
|
4088
|
+
if (editsExpected) {
|
|
4089
|
+
if (hasWorktreeDiff && diffText.length > 0) {
|
|
4090
|
+
passed = true;
|
|
4091
|
+
reasons.push(`diff > 0 chars (${diffText.length})`);
|
|
4092
|
+
}
|
|
4093
|
+
if (filesChanged > 0) {
|
|
4094
|
+
passed = true;
|
|
4095
|
+
reasons.push(`filesChanged > 0 (${filesChanged})`);
|
|
4096
|
+
}
|
|
4097
|
+
} else {
|
|
4098
|
+
if (hasWorktreeDiff && diffText.length > 0) {
|
|
4099
|
+
reasons.push("diff > 0 chars (bonus, editsExpected=false)");
|
|
4100
|
+
}
|
|
4101
|
+
if (filesChanged > 0) {
|
|
4102
|
+
reasons.push("filesChanged > 0 (bonus, editsExpected=false)");
|
|
4103
|
+
}
|
|
4104
|
+
}
|
|
4105
|
+
if (!passed) {
|
|
4106
|
+
reasons.push("No tangible output detected");
|
|
4107
|
+
reasons.push(`stdout: ${cleanStdout.length} chars (min: ${MIN_STDOUT_LENGTH})`);
|
|
4108
|
+
if (editsExpected) {
|
|
4109
|
+
reasons.push(`diff: ${diffText.length} chars`);
|
|
4110
|
+
reasons.push(`filesChanged: ${filesChanged}`);
|
|
4111
|
+
} else {
|
|
4112
|
+
reasons.push("(editsExpected=false: only stdout required)");
|
|
4113
|
+
}
|
|
4114
|
+
}
|
|
4115
|
+
return {
|
|
4116
|
+
passed,
|
|
4117
|
+
reasons,
|
|
4118
|
+
evidence
|
|
4119
|
+
};
|
|
4120
|
+
}
|
|
4121
|
+
function inspectAllVariants(variants, options = {}) {
|
|
4122
|
+
const results = /* @__PURE__ */ new Map();
|
|
4123
|
+
const editsExpected = options?.editsExpected ?? true;
|
|
4124
|
+
for (const variant of variants) {
|
|
4125
|
+
const result = inspectVariantOutput(variant, options);
|
|
4126
|
+
if (!editsExpected && result.evidence.stdoutLength >= MIN_STDOUT_LENGTH && result.evidence.diffLength === 0 && result.evidence.filesChanged === 0) {
|
|
4127
|
+
results.set(variant.variant, {
|
|
4128
|
+
...result,
|
|
4129
|
+
passed: true,
|
|
4130
|
+
reasons: [`stdout >= ${MIN_STDOUT_LENGTH} chars (analysis-only task)`]
|
|
4131
|
+
});
|
|
4132
|
+
} else {
|
|
4133
|
+
results.set(variant.variant, result);
|
|
4134
|
+
}
|
|
4135
|
+
}
|
|
4136
|
+
return results;
|
|
4137
|
+
}
|
|
4138
|
+
|
|
4139
|
+
// src/core/judge/variant-health.ts
|
|
4140
|
+
var STRATEGY_TOKENS = /* @__PURE__ */ new Set([
|
|
4141
|
+
"conservative",
|
|
4142
|
+
"innovative",
|
|
4143
|
+
"hybrid",
|
|
4144
|
+
"fast",
|
|
4145
|
+
"fusion",
|
|
4146
|
+
"balanced",
|
|
4147
|
+
"default"
|
|
4148
|
+
]);
|
|
4149
|
+
var MODE_TOKENS = /* @__PURE__ */ new Set(["local", "cloud", "api", "headless"]);
|
|
4150
|
+
var EFFORT_TOKENS = /* @__PURE__ */ new Set(["minimal", "low", "medium", "high", "extra-high", "xhigh", "max"]);
|
|
4151
|
+
var EFFORT_PREFIX = "effort-";
|
|
4152
|
+
function resolveNoOutputGateMode() {
|
|
4153
|
+
return JUDGE_FEATURE_FLAGS.NO_OUTPUT_PREEVAL_GATE;
|
|
4154
|
+
}
|
|
4155
|
+
function resolveEffectiveNoOutputGateMode(_options) {
|
|
4156
|
+
const configured = resolveNoOutputGateMode();
|
|
4157
|
+
return configured === "off" ? "off" : "shadow";
|
|
4158
|
+
}
|
|
4159
|
+
function normalizeTestProvenanceScope(variant) {
|
|
4160
|
+
const scope = variant.testsExecuted?.scope;
|
|
4161
|
+
if (!scope) return "unknown";
|
|
4162
|
+
if (scope === "none") return "none";
|
|
4163
|
+
if (scope === "full") return "full";
|
|
4164
|
+
const command = variant.testsExecuted?.command?.toLowerCase() ?? "";
|
|
4165
|
+
if (command.includes("smoke")) return "smoke";
|
|
4166
|
+
return "scoped";
|
|
4167
|
+
}
|
|
4168
|
+
function parseVariantIdentity(variant) {
|
|
4169
|
+
const identity = {
|
|
4170
|
+
provider: variant.provider,
|
|
4171
|
+
model: variant.resolvedModel ?? variant.model,
|
|
4172
|
+
strategy: variant.strategy,
|
|
4173
|
+
mode: variant.executionMode
|
|
4174
|
+
};
|
|
4175
|
+
const parts = String(variant.variant ?? "").split(":").filter(Boolean);
|
|
4176
|
+
if (parts.length === 0) return identity;
|
|
4177
|
+
identity.provider ??= parts[0];
|
|
4178
|
+
const last = parts.at(-1);
|
|
4179
|
+
if (last && MODE_TOKENS.has(last)) {
|
|
4180
|
+
identity.mode ??= last;
|
|
4181
|
+
parts.pop();
|
|
4182
|
+
}
|
|
4183
|
+
const maybeStrategy = parts.at(-1);
|
|
4184
|
+
if (maybeStrategy && STRATEGY_TOKENS.has(maybeStrategy)) {
|
|
4185
|
+
identity.strategy ??= maybeStrategy;
|
|
4186
|
+
parts.pop();
|
|
4187
|
+
}
|
|
4188
|
+
const effortPairIndex = parts.findIndex(
|
|
4189
|
+
(part, index) => part === "effort" && EFFORT_TOKENS.has(parts[index + 1] ?? "")
|
|
4190
|
+
);
|
|
4191
|
+
if (effortPairIndex >= 0) {
|
|
4192
|
+
identity.effortLevel ??= parts[effortPairIndex + 1];
|
|
4193
|
+
parts.splice(effortPairIndex, 2);
|
|
4194
|
+
}
|
|
4195
|
+
const effortIndex = parts.findIndex(
|
|
4196
|
+
(part) => part.startsWith(EFFORT_PREFIX) || EFFORT_TOKENS.has(part)
|
|
4197
|
+
);
|
|
4198
|
+
if (effortIndex >= 0) {
|
|
4199
|
+
const effortToken = parts[effortIndex];
|
|
4200
|
+
identity.effortLevel ??= effortToken.startsWith(EFFORT_PREFIX) ? effortToken.slice(EFFORT_PREFIX.length) : effortToken;
|
|
4201
|
+
parts.splice(effortIndex, 1);
|
|
4202
|
+
}
|
|
4203
|
+
if (!identity.model && parts.length >= 2) {
|
|
4204
|
+
identity.model = parts.slice(1).join(":");
|
|
4205
|
+
}
|
|
4206
|
+
return identity;
|
|
4207
|
+
}
|
|
4208
|
+
function classifyVariantHealth(args) {
|
|
4209
|
+
if (isAnalysisIntent(args.variant)) return "valid_analysis";
|
|
4210
|
+
if (args.filesChanged > 0 || args.diffLength > 0) return "valid_materialized";
|
|
4211
|
+
if (args.hasTestExecution) return "valid_test_only";
|
|
4212
|
+
if (args.inspectionPassed) return "weak_evidence";
|
|
4213
|
+
if (args.stdoutLength > 0) return "weak_evidence";
|
|
4214
|
+
return "invalid_materialization";
|
|
4215
|
+
}
|
|
4216
|
+
function buildVariantHealth(input) {
|
|
4217
|
+
const gateMode = input.gateMode ?? resolveNoOutputGateMode();
|
|
4218
|
+
const effectiveGateMode = input.effectiveGateMode ?? gateMode;
|
|
4219
|
+
const cleanStdout = extractCleanStdout(input.variant);
|
|
4220
|
+
const filesChanged = input.variant.gitMetrics?.filesChanged ?? input.variant.filesChanged ?? 0;
|
|
4221
|
+
const stdoutLength = input.inspection?.evidence.stdoutLength ?? cleanStdout.length;
|
|
4222
|
+
const diffLength = input.inspection?.evidence.diffLength ?? (input.variant.worktreeDiff ? extractDiffText(input.variant).length : 0);
|
|
4223
|
+
const hasTestExecution = input.hasTestExecution ?? (input.variant.testsExecuted?.scope ?? "none") !== "none";
|
|
4224
|
+
const inspectionPassed = input.inspection?.passed ?? (stdoutLength > 0 || diffLength > 0 || filesChanged > 0);
|
|
4225
|
+
const classification = classifyVariantHealth({
|
|
4226
|
+
variant: input.variant,
|
|
4227
|
+
filesChanged,
|
|
4228
|
+
stdoutLength,
|
|
4229
|
+
diffLength,
|
|
4230
|
+
hasTestExecution,
|
|
4231
|
+
inspectionPassed
|
|
4232
|
+
});
|
|
4233
|
+
const wouldExclude = input.wouldExclude ?? false;
|
|
4234
|
+
const wasExcluded = false;
|
|
4235
|
+
const identity = parseVariantIdentity(input.variant);
|
|
4236
|
+
return {
|
|
4237
|
+
taskId: input.options?.taskId,
|
|
4238
|
+
judgeKind: input.judgeKind,
|
|
4239
|
+
gateMode,
|
|
4240
|
+
effectiveGateMode,
|
|
4241
|
+
judgeStrict: input.options?.judgeStrict === true,
|
|
4242
|
+
variant: input.variant.variant,
|
|
4243
|
+
provider: identity.provider,
|
|
4244
|
+
model: identity.model,
|
|
4245
|
+
strategy: identity.strategy,
|
|
4246
|
+
effortLevel: identity.effortLevel,
|
|
4247
|
+
mode: identity.mode,
|
|
4248
|
+
promptIntent: input.promptIntent,
|
|
4249
|
+
executionIntent: input.variant.executionIntent,
|
|
4250
|
+
filesChanged,
|
|
4251
|
+
diffLength,
|
|
4252
|
+
stdoutLength,
|
|
4253
|
+
testProvenanceScope: normalizeTestProvenanceScope(input.variant),
|
|
4254
|
+
classification,
|
|
4255
|
+
shownToLlm: input.shownToLlm ?? !wasExcluded,
|
|
4256
|
+
candidateSetImpact: {
|
|
4257
|
+
wouldExclude,
|
|
4258
|
+
wasExcluded,
|
|
4259
|
+
counterfactualWinnerChange: "unknown"
|
|
4260
|
+
}
|
|
4261
|
+
};
|
|
4262
|
+
}
|
|
4263
|
+
function logVariantHealthEvents(telemetry, variantHealth) {
|
|
4264
|
+
if (!telemetry?.logVariantHealth) return;
|
|
4265
|
+
for (const health of variantHealth) {
|
|
4266
|
+
telemetry.logVariantHealth(health);
|
|
4267
|
+
}
|
|
4268
|
+
}
|
|
4269
|
+
function observeVariantHealth(args) {
|
|
4270
|
+
const r5AnalysisExempt = args.explicitReviewExemptsR5 && args.isExplicitReview;
|
|
4271
|
+
const noOutputGateMode = resolveNoOutputGateMode();
|
|
4272
|
+
const effectiveNoOutputGateMode = resolveEffectiveNoOutputGateMode(args.options);
|
|
4273
|
+
const variantHealth = args.variants.map((variant) => {
|
|
4274
|
+
const inspection = args.inspectionLookup?.(variant) ?? inspectVariantOutput(variant);
|
|
4275
|
+
const hasTestExecution = hasTestExecutionProvenance(variant);
|
|
4276
|
+
const wouldExclude = !r5AnalysisExempt && !inspection.passed && !hasTestExecution && !isAnalysisIntent(variant);
|
|
4277
|
+
return buildVariantHealth({
|
|
4278
|
+
variant,
|
|
4279
|
+
judgeKind: args.judgeKind,
|
|
4280
|
+
promptIntent: args.isExplicitReview ? "explicit-review" : void 0,
|
|
4281
|
+
options: args.options,
|
|
4282
|
+
inspection,
|
|
4283
|
+
hasTestExecution,
|
|
4284
|
+
gateMode: noOutputGateMode,
|
|
4285
|
+
effectiveGateMode: effectiveNoOutputGateMode,
|
|
4286
|
+
wouldExclude,
|
|
4287
|
+
wasExcluded: false,
|
|
4288
|
+
shownToLlm: true
|
|
4289
|
+
});
|
|
4290
|
+
});
|
|
4291
|
+
logVariantHealthEvents(args.telemetry, variantHealth);
|
|
4292
|
+
logR5NoOutputShadowObservations({
|
|
4293
|
+
gateMode: noOutputGateMode,
|
|
4294
|
+
variantHealth,
|
|
4295
|
+
taskId: args.taskId,
|
|
4296
|
+
telemetry: args.telemetry,
|
|
4297
|
+
resolveVariantId: args.resolveVariantId,
|
|
4298
|
+
onWouldExclude: args.onWouldExclude
|
|
4299
|
+
});
|
|
4300
|
+
return variantHealth;
|
|
4301
|
+
}
|
|
4302
|
+
function logR5NoOutputShadowObservations(args) {
|
|
4303
|
+
if (args.gateMode === "off") {
|
|
4304
|
+
return;
|
|
4305
|
+
}
|
|
4306
|
+
const wouldExcludeHealth = args.variantHealth.filter(
|
|
4307
|
+
(health) => health.candidateSetImpact.wouldExclude
|
|
4308
|
+
);
|
|
4309
|
+
if (wouldExcludeHealth.length === 0) {
|
|
4310
|
+
return;
|
|
4311
|
+
}
|
|
4312
|
+
const resolveVariantId = args.resolveVariantId ?? ((variantId) => variantId);
|
|
4313
|
+
for (const health of wouldExcludeHealth) {
|
|
4314
|
+
args.onWouldExclude?.(resolveVariantId(health.variant));
|
|
4315
|
+
}
|
|
4316
|
+
const taskId = args.taskId?.trim();
|
|
4317
|
+
if (!taskId || !args.telemetry?.logShadowDelta) {
|
|
4318
|
+
return;
|
|
4319
|
+
}
|
|
4320
|
+
args.telemetry.logShadowDelta({
|
|
4321
|
+
tier: "gate",
|
|
4322
|
+
flag: "NO_OUTPUT_PREEVAL_GATE",
|
|
4323
|
+
taskId,
|
|
4324
|
+
applied: false,
|
|
4325
|
+
wouldChangeEligibility: true,
|
|
4326
|
+
affectedVariants: wouldExcludeHealth.map((health) => ({
|
|
4327
|
+
variant: resolveVariantId(health.variant),
|
|
4328
|
+
currentEligible: true,
|
|
4329
|
+
shadowEligible: false
|
|
4330
|
+
}))
|
|
4331
|
+
});
|
|
4332
|
+
}
|
|
4333
|
+
|
|
4334
|
+
// src/core/judge/winner-from-scores.ts
|
|
4335
|
+
var DEFAULT_DETERMINISTIC_WINNER_WEIGHTS = {
|
|
4336
|
+
delivery: DETERMINISTIC_WINNER_WEIGHTS.delivery,
|
|
4337
|
+
correctness: DETERMINISTIC_WINNER_WEIGHTS.correctness,
|
|
4338
|
+
quality: DETERMINISTIC_WINNER_WEIGHTS.quality
|
|
4339
|
+
};
|
|
4340
|
+
var DEFAULT_DETERMINISTIC_WINNER_THRESHOLDS = {
|
|
4341
|
+
abstainMargin: 0.5,
|
|
4342
|
+
tieMargin: 2
|
|
4343
|
+
};
|
|
4344
|
+
function toFiniteNumber(value) {
|
|
4345
|
+
const n = Number(value);
|
|
4346
|
+
return Number.isFinite(n) ? n : null;
|
|
4347
|
+
}
|
|
4348
|
+
function clampScore(score, outOfRange) {
|
|
4349
|
+
if (score < 0 || score > 100) {
|
|
4350
|
+
if (outOfRange) outOfRange.count++;
|
|
4351
|
+
}
|
|
4352
|
+
return Math.max(0, Math.min(100, score));
|
|
4353
|
+
}
|
|
4354
|
+
function weighted(delivery, correctness, quality, weights) {
|
|
4355
|
+
return delivery * weights.delivery + correctness * weights.correctness + quality * weights.quality;
|
|
4356
|
+
}
|
|
4357
|
+
function scoreVariant(entry, weights, outOfRange) {
|
|
4358
|
+
const declaredOverallScore = entry.declaredOverallScore;
|
|
4359
|
+
if (typeof declaredOverallScore === "number" && Number.isFinite(declaredOverallScore) && declaredOverallScore >= 0 && declaredOverallScore <= 100) {
|
|
4360
|
+
return {
|
|
4361
|
+
score: declaredOverallScore,
|
|
4362
|
+
source: "llm_declared_overall_score"
|
|
4363
|
+
};
|
|
4364
|
+
}
|
|
4365
|
+
const bucketDelivery = toFiniteNumber(entry.bucketScores?.delivery);
|
|
4366
|
+
const bucketCorrectness = toFiniteNumber(entry.bucketScores?.correctness);
|
|
4367
|
+
const bucketQuality = toFiniteNumber(entry.bucketScores?.quality);
|
|
4368
|
+
if (bucketDelivery != null && bucketCorrectness != null && bucketQuality != null) {
|
|
4369
|
+
return {
|
|
4370
|
+
score: clampScore(
|
|
4371
|
+
weighted(bucketDelivery, bucketCorrectness, bucketQuality, weights),
|
|
4372
|
+
outOfRange
|
|
4373
|
+
),
|
|
4374
|
+
source: "bucket_scores"
|
|
4375
|
+
};
|
|
4376
|
+
}
|
|
4377
|
+
const codeCorrectness = toFiniteNumber(entry.codeQuality?.correctness);
|
|
4378
|
+
const codeQualitySignals = [
|
|
4379
|
+
toFiniteNumber(entry.codeQuality?.readability),
|
|
4380
|
+
toFiniteNumber(entry.codeQuality?.maintainability),
|
|
4381
|
+
toFiniteNumber(entry.codeQuality?.innovation)
|
|
4382
|
+
].filter((value) => value != null);
|
|
4383
|
+
const codeQuality = codeQualitySignals.length > 0 ? codeQualitySignals.reduce((sum, value) => sum + value, 0) / codeQualitySignals.length : null;
|
|
4384
|
+
const codeCompleteness = toFiniteNumber(entry.codeQuality?.completeness);
|
|
4385
|
+
if (codeCompleteness != null && codeCorrectness != null && codeQuality != null) {
|
|
4386
|
+
return {
|
|
4387
|
+
score: clampScore(
|
|
4388
|
+
weighted(codeCompleteness, codeCorrectness, codeQuality, weights),
|
|
4389
|
+
outOfRange
|
|
4390
|
+
),
|
|
4391
|
+
source: "code_quality"
|
|
4392
|
+
};
|
|
4393
|
+
}
|
|
4394
|
+
const qualityScore = toFiniteNumber(entry.qualityScore);
|
|
4395
|
+
if (qualityScore != null) {
|
|
4396
|
+
return { score: clampScore(qualityScore, outOfRange), source: "quality_score" };
|
|
4397
|
+
}
|
|
4398
|
+
return { score: 0, source: "fallback_zero" };
|
|
4399
|
+
}
|
|
4400
|
+
function normalizeThresholds(thresholds) {
|
|
4401
|
+
const abstainMargin = Math.max(
|
|
4402
|
+
0,
|
|
4403
|
+
Number.isFinite(Number(thresholds?.abstainMargin)) ? Number(thresholds?.abstainMargin) : DEFAULT_DETERMINISTIC_WINNER_THRESHOLDS.abstainMargin
|
|
4404
|
+
);
|
|
4405
|
+
const tieCandidate = Math.max(
|
|
4406
|
+
0,
|
|
4407
|
+
Number.isFinite(Number(thresholds?.tieMargin)) ? Number(thresholds?.tieMargin) : DEFAULT_DETERMINISTIC_WINNER_THRESHOLDS.tieMargin
|
|
4408
|
+
);
|
|
4409
|
+
return {
|
|
4410
|
+
abstainMargin,
|
|
4411
|
+
tieMargin: Math.max(tieCandidate, abstainMargin)
|
|
4412
|
+
};
|
|
4413
|
+
}
|
|
4414
|
+
function selectDeterministicWinnerFromScores(entries, options) {
|
|
4415
|
+
if (!entries || entries.length === 0) {
|
|
4416
|
+
return {
|
|
4417
|
+
winner: null,
|
|
4418
|
+
reason: "no_variants",
|
|
4419
|
+
margin: null,
|
|
4420
|
+
top2: [],
|
|
4421
|
+
scored: [],
|
|
4422
|
+
thresholds: normalizeThresholds(options?.thresholds),
|
|
4423
|
+
outOfRangeCount: 0
|
|
4424
|
+
};
|
|
4425
|
+
}
|
|
4426
|
+
const weights = {
|
|
4427
|
+
delivery: Number.isFinite(Number(options?.weights?.delivery)) ? Number(options?.weights?.delivery) : DEFAULT_DETERMINISTIC_WINNER_WEIGHTS.delivery,
|
|
4428
|
+
correctness: Number.isFinite(Number(options?.weights?.correctness)) ? Number(options?.weights?.correctness) : DEFAULT_DETERMINISTIC_WINNER_WEIGHTS.correctness,
|
|
4429
|
+
quality: Number.isFinite(Number(options?.weights?.quality)) ? Number(options?.weights?.quality) : DEFAULT_DETERMINISTIC_WINNER_WEIGHTS.quality
|
|
4430
|
+
};
|
|
4431
|
+
const thresholds = normalizeThresholds(options?.thresholds);
|
|
4432
|
+
const outOfRange = { count: 0 };
|
|
4433
|
+
const scored = entries.map((entry) => {
|
|
4434
|
+
const scoredVariant = scoreVariant(entry, weights, outOfRange);
|
|
4435
|
+
return {
|
|
4436
|
+
variant: entry.variant,
|
|
4437
|
+
compositeScore: createPercentScore(
|
|
4438
|
+
scoredVariant.score,
|
|
4439
|
+
`deterministic score ${entry.variant}`
|
|
4440
|
+
),
|
|
4441
|
+
scoreSource: scoredVariant.source
|
|
4442
|
+
};
|
|
4443
|
+
});
|
|
4444
|
+
scored.sort((a, b) => {
|
|
4445
|
+
const scoreDelta = Number(b.compositeScore) - Number(a.compositeScore);
|
|
4446
|
+
if (scoreDelta !== 0) return scoreDelta;
|
|
4447
|
+
return a.variant.localeCompare(b.variant);
|
|
4448
|
+
});
|
|
4449
|
+
const eligibleScored = scored;
|
|
4450
|
+
const leader = eligibleScored[0];
|
|
4451
|
+
const runnerUp = eligibleScored[1];
|
|
4452
|
+
const margin = leader && runnerUp ? Number(leader.compositeScore) - Number(runnerUp.compositeScore) : null;
|
|
4453
|
+
const top2 = eligibleScored.slice(0, 2).map((item) => ({
|
|
4454
|
+
variant: item.variant,
|
|
4455
|
+
score: Number(item.compositeScore)
|
|
4456
|
+
}));
|
|
4457
|
+
if (!leader) {
|
|
4458
|
+
return {
|
|
4459
|
+
winner: null,
|
|
4460
|
+
reason: "no_variants",
|
|
4461
|
+
margin,
|
|
4462
|
+
top2,
|
|
4463
|
+
scored,
|
|
4464
|
+
thresholds,
|
|
4465
|
+
outOfRangeCount: outOfRange.count
|
|
4466
|
+
};
|
|
4467
|
+
}
|
|
4468
|
+
const isTiebreak = runnerUp != null && margin === 0;
|
|
4469
|
+
return {
|
|
4470
|
+
winner: leader.variant,
|
|
4471
|
+
reason: "winner_by_score",
|
|
4472
|
+
margin,
|
|
4473
|
+
top2,
|
|
4474
|
+
scored,
|
|
4475
|
+
thresholds,
|
|
4476
|
+
outOfRangeCount: outOfRange.count,
|
|
4477
|
+
...isTiebreak ? { tiebreakReason: "lexical_name_order" } : {}
|
|
4478
|
+
};
|
|
4479
|
+
}
|
|
4480
|
+
function computeSelfConsistencyWarning(winnerId, deterministicWinner) {
|
|
4481
|
+
if (deterministicWinner.scored.length === 0) return void 0;
|
|
4482
|
+
const topScorer = deterministicWinner.scored[0];
|
|
4483
|
+
const winnerScored = deterministicWinner.scored.find((s) => s.variant === winnerId);
|
|
4484
|
+
if (topScorer && winnerScored && topScorer.variant !== winnerId && Number(topScorer.compositeScore) - Number(winnerScored.compositeScore) > 10) {
|
|
4485
|
+
return `Ranking winner "${winnerId}" composite=${Number(winnerScored.compositeScore).toFixed(1)} is >10pts below top scorer "${topScorer.variant}" composite=${Number(topScorer.compositeScore).toFixed(1)}`;
|
|
4486
|
+
}
|
|
4487
|
+
return void 0;
|
|
4488
|
+
}
|
|
4489
|
+
|
|
4490
|
+
export {
|
|
4491
|
+
isCodeIntent,
|
|
4492
|
+
runPreScoreGates,
|
|
4493
|
+
createPercentScore,
|
|
4494
|
+
percentScoreToNumber,
|
|
4495
|
+
scanDiffForStructuralRisks,
|
|
4496
|
+
evaluateACComplianceItems,
|
|
4497
|
+
buildACComplianceSection,
|
|
4498
|
+
requiresInputSimulationCheck,
|
|
4499
|
+
computePromptOrderEvidenceCandidateLimit,
|
|
4500
|
+
computeAdaptiveJudgePromptBudgetCaps,
|
|
4501
|
+
buildVariantEvidenceCandidatesForJudge,
|
|
4502
|
+
evidenceQuoteIsValid,
|
|
4503
|
+
buildVariantDeliverableExcerptForJudge,
|
|
4504
|
+
resolveAcceptanceCriteriaTargets,
|
|
4505
|
+
buildAIPrompt3Bucket,
|
|
4506
|
+
SeededRandom,
|
|
4507
|
+
createFallbackAnonymizationSeed,
|
|
4508
|
+
anonymizeVariants,
|
|
4509
|
+
restoreOriginalOrder,
|
|
4510
|
+
resolveAnonymizationRunId,
|
|
4511
|
+
shouldAnonymize,
|
|
4512
|
+
createAnonymizedVariantCopies,
|
|
4513
|
+
deAnonymizeRationale,
|
|
4514
|
+
detectDuplicateContent,
|
|
4515
|
+
persistJudgeFailureArtifact,
|
|
4516
|
+
getDefaultJudgeModel,
|
|
4517
|
+
getDefaultMultiLensJudgeModel,
|
|
4518
|
+
normalizeJudgeModel,
|
|
4519
|
+
resolveBuildGatePromptSignal,
|
|
4520
|
+
resolveLintGatePromptStatus,
|
|
4521
|
+
resolveTestGatePromptSource,
|
|
4522
|
+
computeDeterministicPreScoreGateContext,
|
|
4523
|
+
buildDeterministicPreScoreGateSection,
|
|
4524
|
+
normalizeRankingIds,
|
|
4525
|
+
validateRanking,
|
|
4526
|
+
checkRankingScoreConsistency,
|
|
4527
|
+
inspectVariantOutput,
|
|
4528
|
+
inspectAllVariants,
|
|
4529
|
+
observeVariantHealth,
|
|
4530
|
+
selectDeterministicWinnerFromScores,
|
|
4531
|
+
computeSelfConsistencyWarning
|
|
4532
|
+
};
|
|
4533
|
+
//# sourceMappingURL=chunk-TSPQONRT.js.map
|