@rulemetric/cli 0.7.3 → 0.7.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -8
- package/dist/base-command.js +1 -14
- package/dist/chunk-2JRDG4LJ.js +31 -0
- package/dist/chunk-2TC6WCUT.js +34 -0
- package/dist/chunk-2WLVDSAP.js +2 -0
- package/dist/chunk-32EEBARV.js +1 -0
- package/dist/chunk-3AV3GTMP.js +1 -0
- package/dist/chunk-43YBXX7R.js +1 -0
- package/dist/chunk-4Z5ECCJE.js +1 -0
- package/dist/chunk-5RBLIVBF.js +1 -0
- package/dist/chunk-67QCYUJU.js +1 -0
- package/dist/chunk-6QKI756T.js +2 -0
- package/dist/chunk-6V4WM73L.js +1 -0
- package/dist/chunk-6XEC6474.js +1 -0
- package/dist/chunk-7DFNBRLY.js +61 -0
- package/dist/chunk-ARNZQBMX.js +1 -0
- package/dist/chunk-AYUJY37O.js +5 -0
- package/dist/chunk-CADKBP3D.js +58 -0
- package/dist/chunk-CDI5CCHU.js +1 -0
- package/dist/chunk-CNQ5DTNC.js +1 -0
- package/dist/chunk-DCLWFGEA.js +1 -0
- package/dist/chunk-DNF4HDI6.js +14 -0
- package/dist/chunk-DPFNJZUO.js +1 -0
- package/dist/chunk-DSO2U4Y6.js +56 -0
- package/dist/chunk-DVN7GFJG.js +1 -0
- package/dist/chunk-G63KMH6U.js +1 -0
- package/dist/chunk-G6XGDHGC.js +2 -0
- package/dist/chunk-GBMF3TP6.js +1 -0
- package/dist/chunk-GY5IJ4RS.js +1 -0
- package/dist/chunk-H2ORWIEY.js +1 -0
- package/dist/chunk-H4SV7FN4.js +1 -0
- package/dist/chunk-HS7X43XE.js +24 -0
- package/dist/{chunk-OQBBVTDG.js → chunk-I2DZGJRR.js} +6 -71
- package/dist/chunk-IZTIYR5Q.js +2 -0
- package/dist/chunk-JAPCCEKF.js +3 -0
- package/dist/chunk-JQTGQXUX.js +1 -0
- package/dist/chunk-KAF3CYRW.js +9 -0
- package/dist/chunk-KKDHD4XA.js +2 -0
- package/dist/chunk-KV325CXF.js +33 -0
- package/dist/chunk-MGHFDHHM.js +1 -0
- package/dist/chunk-MODXEGCU.js +1 -0
- package/dist/chunk-MVM6OGAF.js +4 -0
- package/dist/chunk-N7ODX6PO.js +1 -0
- package/dist/chunk-N7TGRFVZ.js +70 -0
- package/dist/chunk-NBTC5FNV.js +1 -0
- package/dist/chunk-NEACEHQD.js +1 -0
- package/dist/chunk-NJ5UED3H.js +1 -0
- package/dist/chunk-NVKX36YN.js +3 -0
- package/dist/chunk-O2SDD2WW.js +1 -0
- package/dist/chunk-O6C2UGBY.js +7 -0
- package/dist/chunk-OMXUK4M6.js +1 -0
- package/dist/chunk-PPKPCXJD.js +4 -0
- package/dist/{chunk-XZXS2W24.js → chunk-QFVGUZDO.js} +11 -57
- package/dist/chunk-RDET7CKR.js +1 -0
- package/dist/chunk-RQ424GI6.js +5 -0
- package/dist/chunk-RSKQRVPJ.js +1 -0
- package/dist/chunk-RTWRHIKY.js +7 -0
- package/dist/chunk-SZ62CP55.js +1 -0
- package/dist/chunk-TDD7IOV2.js +1 -0
- package/dist/chunk-TIOKY2IA.js +1 -0
- package/dist/chunk-TQMG647B.js +15 -0
- package/dist/chunk-TU2AR6XO.js +1 -0
- package/dist/chunk-U6ABMRT3.js +9 -0
- package/dist/chunk-U6XEU573.js +1 -0
- package/dist/chunk-UF6US7VU.js +1 -0
- package/dist/chunk-V44SRRYR.js +1 -0
- package/dist/chunk-V5U3E4BZ.js +1 -0
- package/dist/chunk-VAXRCSSH.js +2 -0
- package/dist/chunk-VLOX6KOM.js +1 -0
- package/dist/chunk-VUGRXM7R.js +1 -0
- package/dist/chunk-W2MOTOQE.js +1 -0
- package/dist/chunk-W2MV25SB.js +1 -0
- package/dist/chunk-WLVQJH4E.js +1 -0
- package/dist/chunk-WSCK4WNK.js +1 -0
- package/dist/chunk-XICLXXRH.js +1 -0
- package/dist/chunk-XXDZJ6YI.js +1 -0
- package/dist/chunk-ZL4MZMUK.js +34 -0
- package/dist/chunk-ZLLER3HP.js +1 -0
- package/dist/chunk-ZR3MUNLS.js +1 -0
- package/dist/chunk-ZXF5NE54.js +1 -0
- package/dist/commands/antigravity/backfill.js +2 -136
- package/dist/commands/auth/create-key.js +1 -52
- package/dist/commands/auth/login.js +1 -85
- package/dist/commands/auth/logout.js +1 -29
- package/dist/commands/auth/revoke-key.js +4 -33
- package/dist/commands/auth/signup.js +1 -89
- package/dist/commands/auth/status.js +1 -42
- package/dist/commands/catalog/index.js +2 -52
- package/dist/commands/catalog/recommend.js +3 -43
- package/dist/commands/convert.js +1 -75
- package/dist/commands/converters.js +1 -54
- package/dist/commands/crons/list.js +1 -41
- package/dist/commands/crons/run.js +2 -76
- package/dist/commands/dashboard.js +1 -46
- package/dist/commands/diagnostics.js +1 -117
- package/dist/commands/eval-targets/create.js +1 -53
- package/dist/commands/eval-targets/list.js +1 -54
- package/dist/commands/evals/agent.js +3 -310
- package/dist/commands/evals/benchmark.js +6 -73
- package/dist/commands/evals/compare.js +13 -96
- package/dist/commands/evals/create.js +1 -67
- package/dist/commands/evals/optimize-description.js +25 -239
- package/dist/commands/evals/optimize.js +5 -211
- package/dist/commands/evals/run.js +3 -183
- package/dist/commands/gateway/ensure.js +1 -104
- package/dist/commands/hooks/emit-instructions.js +1 -25
- package/dist/commands/hooks/install.js +1 -442
- package/dist/commands/hooks/run.js +3 -81
- package/dist/commands/hooks/uninstall.js +2 -156
- package/dist/commands/import.js +1 -66
- package/dist/commands/instructions/create.js +1 -74
- package/dist/commands/instructions/delete.js +1 -55
- package/dist/commands/instructions/fork.js +1 -34
- package/dist/commands/instructions/get.js +1 -68
- package/dist/commands/instructions/list.js +1 -65
- package/dist/commands/instructions/promote.js +1 -42
- package/dist/commands/instructions/pull.js +1 -39
- package/dist/commands/instructions/upstream.js +1 -38
- package/dist/commands/instructions/versions.js +1 -53
- package/dist/commands/launcher.js +1 -40
- package/dist/commands/org/current.js +1 -77
- package/dist/commands/org/list.js +1 -52
- package/dist/commands/org/switch.js +1 -71
- package/dist/commands/packs/add-instruction.js +1 -37
- package/dist/commands/packs/apply.js +1 -39
- package/dist/commands/packs/create.js +1 -50
- package/dist/commands/packs/fork.js +1 -34
- package/dist/commands/packs/get.js +3 -56
- package/dist/commands/packs/list.js +1 -66
- package/dist/commands/packs/remove-instruction.js +1 -35
- package/dist/commands/proxy/env.js +3 -131
- package/dist/commands/proxy/logs.js +3 -85
- package/dist/commands/proxy/restart.js +1 -35
- package/dist/commands/proxy/setup.js +17 -454
- package/dist/commands/proxy/start.js +10 -333
- package/dist/commands/proxy/status.js +1 -78
- package/dist/commands/proxy/stop.js +1 -94
- package/dist/commands/research/backfill.js +1 -59
- package/dist/commands/research/tail-transcript.js +1 -70
- package/dist/commands/service/install.js +57 -796
- package/dist/commands/service/status.js +1 -180
- package/dist/commands/service/uninstall.js +1 -119
- package/dist/commands/sessions/analyze.js +1 -48
- package/dist/commands/sessions/end.js +2 -85
- package/dist/commands/sessions/import.js +1 -46
- package/dist/commands/sessions/list.js +1 -48
- package/dist/commands/sessions/start.js +3 -90
- package/dist/commands/setup.js +3 -150
- package/dist/commands/skills/export-approved.js +1 -46
- package/dist/commands/skills/install.js +4 -107
- package/dist/commands/skills/list.js +4 -65
- package/dist/commands/skills/publish.js +2 -98
- package/dist/commands/skills/search.js +3 -72
- package/dist/commands/skills/update.js +3 -55
- package/dist/commands/suggestions/accept.js +8 -84
- package/dist/commands/suggestions/dismiss.js +1 -45
- package/dist/commands/suggestions/list.js +2 -61
- package/dist/commands/suggestions/refresh.js +1 -27
- package/dist/cursor-T7AMCMKS.js +13 -0
- package/dist/dashboard/Dashboard.js +1 -243
- package/dist/dashboard/data.js +1 -15
- package/dist/dashboard/launcher/ActiveJobsList.js +1 -10
- package/dist/dashboard/launcher/LauncherOverlay.js +1 -16
- package/dist/dashboard/launcher/NewLaunchForm.js +1 -10
- package/dist/dashboard/launcher/api.js +1 -16
- package/dist/dashboard/launcher/tool-labels.js +1 -10
- package/dist/dashboard/launcher/types.js +0 -1
- package/dist/dashboard/launcher/useLaunchJobs.js +1 -11
- package/dist/dashboard/panels/friction.js +1 -10
- package/dist/dashboard/panels/insights.js +1 -10
- package/dist/dashboard/panels/instructions.js +1 -10
- package/dist/dashboard/panels/outcomes.js +1 -10
- package/dist/dashboard/panels/overview.js +1 -10
- package/dist/dashboard/panels/shared.js +1 -19
- package/dist/dashboard/types.js +1 -24
- package/dist/dist-FAVALJ5C.js +1 -0
- package/dist/index.js +1 -8
- package/dist/lib/active-org-refresh.js +1 -15
- package/dist/lib/active-org.js +1 -16
- package/dist/lib/agent-loop.js +1 -47
- package/dist/lib/antigravity-transcript.js +1 -22
- package/dist/lib/api-client.js +1 -23
- package/dist/lib/auth.js +1 -30
- package/dist/lib/capture-hosts.js +1 -8
- package/dist/lib/detect-languages.js +1 -8
- package/dist/lib/detect-tmux.js +1 -9
- package/dist/lib/diagnostics.js +1 -12
- package/dist/lib/editor-proxy-pins.js +1 -14
- package/dist/lib/ensure-api-key.js +1 -11
- package/dist/lib/ensure-fresh-token.js +1 -0
- package/dist/lib/eval-analyzer.js +1 -9
- package/dist/lib/eval-benchmark.js +1 -12
- package/dist/lib/eval-conversation-handler.js +1 -13
- package/dist/lib/eval-executor.js +1 -8
- package/dist/lib/eval-grader.js +1 -9
- package/dist/lib/eval-meta-judge.js +1 -9
- package/dist/lib/eval-trigger-tester.js +1 -11
- package/dist/lib/eval-worktree.js +1 -8
- package/dist/lib/gateway-entry.js +1 -61
- package/dist/lib/gateway-identity.js +1 -16
- package/dist/lib/gateway-lifecycle.js +1 -32
- package/dist/lib/gateway.js +1 -14
- package/dist/lib/handlers/cron-refresh-pricing.js +1 -0
- package/dist/lib/handlers/cron-refresh-skills.js +1 -8
- package/dist/lib/handlers/cron-suggest-instructions.js +1 -16
- package/dist/lib/handlers/process-announcement.js +1 -11
- package/dist/lib/handlers/process-changelog.js +1 -11
- package/dist/lib/handlers/process-conversation.js +1 -14
- package/dist/lib/handlers/process-eval.js +1 -15
- package/dist/lib/handlers/process-insights.js +1 -15
- package/dist/lib/handlers/process-launch.js +1 -16
- package/dist/lib/handlers/process-run-cleanup.js +1 -13
- package/dist/lib/handlers/process-send-message.js +1 -10
- package/dist/lib/handlers/process-session-goal.js +1 -12
- package/dist/lib/hooks-config.js +1 -81
- package/dist/lib/insights-prompts.js +1 -12
- package/dist/lib/instruction-snapshot.js +1 -8
- package/dist/lib/job-helpers.js +0 -1
- package/dist/lib/llm-client.js +1 -20
- package/dist/lib/manual-tasks-meta.js +1 -8
- package/dist/lib/manual-tasks.js +1 -80
- package/dist/lib/output.js +1 -12
- package/dist/lib/permanent-job-error.js +1 -10
- package/dist/lib/pid-identity.js +1 -14
- package/dist/lib/proxy-respawn.js +1 -12
- package/dist/lib/research-client.js +1 -12
- package/dist/lib/setup-steps.js +1 -33
- package/dist/lib/skills-loader.js +1 -8
- package/dist/lib/statusline-shim.js +1 -20
- package/dist/lib/telemetry.js +1 -16
- package/dist/lib/terminal-target-heartbeat.js +1 -15
- package/dist/lib/tmux.js +1 -16
- package/dist/lib/transcript-rate-limit-watcher.js +1 -22
- package/dist/lib/types.js +1 -8
- package/dist/lib/version-check.js +1 -12
- package/dist/lib/which.js +1 -8
- package/dist/lib/worktree.js +1 -17
- package/dist/opencode-QRJ3BH23.js +1 -0
- package/oclif.manifest.json +12 -34
- package/package.json +9 -8
- package/dist/base-command.js.map +0 -7
- package/dist/chunk-2HK7RGZB.js +0 -82
- package/dist/chunk-2HK7RGZB.js.map +0 -7
- package/dist/chunk-2ORJWTNZ.js +0 -307
- package/dist/chunk-2ORJWTNZ.js.map +0 -7
- package/dist/chunk-37LN5O7M.js +0 -59
- package/dist/chunk-37LN5O7M.js.map +0 -7
- package/dist/chunk-3S7223UR.js +0 -81
- package/dist/chunk-3S7223UR.js.map +0 -7
- package/dist/chunk-3TIMQ3O6.js +0 -96
- package/dist/chunk-3TIMQ3O6.js.map +0 -7
- package/dist/chunk-42GFSAJP.js +0 -32
- package/dist/chunk-42GFSAJP.js.map +0 -7
- package/dist/chunk-4BHDHHKL.js +0 -48
- package/dist/chunk-4BHDHHKL.js.map +0 -7
- package/dist/chunk-4D537GXE.js +0 -99
- package/dist/chunk-4D537GXE.js.map +0 -7
- package/dist/chunk-4LYCDS2N.js +0 -139
- package/dist/chunk-4LYCDS2N.js.map +0 -7
- package/dist/chunk-4SCZK7AL.js +0 -129
- package/dist/chunk-4SCZK7AL.js.map +0 -7
- package/dist/chunk-4XAWK6S2.js +0 -161
- package/dist/chunk-4XAWK6S2.js.map +0 -7
- package/dist/chunk-52WIYRZH.js +0 -73
- package/dist/chunk-52WIYRZH.js.map +0 -7
- package/dist/chunk-5BAOGNUH.js +0 -79
- package/dist/chunk-5BAOGNUH.js.map +0 -7
- package/dist/chunk-5EDMGK3W.js +0 -39
- package/dist/chunk-5EDMGK3W.js.map +0 -7
- package/dist/chunk-6NBS5XFS.js +0 -99
- package/dist/chunk-6NBS5XFS.js.map +0 -7
- package/dist/chunk-7LVVKQMZ.js +0 -98
- package/dist/chunk-7LVVKQMZ.js.map +0 -7
- package/dist/chunk-7NNFYWKK.js +0 -243
- package/dist/chunk-7NNFYWKK.js.map +0 -7
- package/dist/chunk-7NYKWN7U.js +0 -12
- package/dist/chunk-7NYKWN7U.js.map +0 -7
- package/dist/chunk-BO76WKJR.js +0 -42
- package/dist/chunk-BO76WKJR.js.map +0 -7
- package/dist/chunk-BRD5RUGM.js +0 -53
- package/dist/chunk-BRD5RUGM.js.map +0 -7
- package/dist/chunk-BT6Z6545.js +0 -43
- package/dist/chunk-BT6Z6545.js.map +0 -7
- package/dist/chunk-BZDKEQKU.js +0 -39
- package/dist/chunk-BZDKEQKU.js.map +0 -7
- package/dist/chunk-DGHWRQXL.js +0 -17
- package/dist/chunk-DGHWRQXL.js.map +0 -7
- package/dist/chunk-DIQUV5FR.js +0 -257
- package/dist/chunk-DIQUV5FR.js.map +0 -7
- package/dist/chunk-DMX5YOTM.js +0 -167
- package/dist/chunk-DMX5YOTM.js.map +0 -7
- package/dist/chunk-DNKSVWMD.js +0 -108
- package/dist/chunk-DNKSVWMD.js.map +0 -7
- package/dist/chunk-DQ3KELXD.js +0 -74
- package/dist/chunk-DQ3KELXD.js.map +0 -7
- package/dist/chunk-DUJSRBLP.js +0 -122
- package/dist/chunk-DUJSRBLP.js.map +0 -7
- package/dist/chunk-E3BIT53W.js +0 -154
- package/dist/chunk-E3BIT53W.js.map +0 -7
- package/dist/chunk-EARWG4EV.js +0 -22
- package/dist/chunk-EARWG4EV.js.map +0 -7
- package/dist/chunk-EI5BHWXA.js +0 -1662
- package/dist/chunk-EI5BHWXA.js.map +0 -7
- package/dist/chunk-EJ3YAFDP.js +0 -37
- package/dist/chunk-EJ3YAFDP.js.map +0 -7
- package/dist/chunk-EKJ2CABV.js +0 -76
- package/dist/chunk-EKJ2CABV.js.map +0 -7
- package/dist/chunk-EKP32DLN.js +0 -16
- package/dist/chunk-EKP32DLN.js.map +0 -7
- package/dist/chunk-EYPAAGN5.js +0 -51
- package/dist/chunk-EYPAAGN5.js.map +0 -7
- package/dist/chunk-F5N6CLJJ.js +0 -1995
- package/dist/chunk-F5N6CLJJ.js.map +0 -7
- package/dist/chunk-GIUBARWS.js +0 -99
- package/dist/chunk-GIUBARWS.js.map +0 -7
- package/dist/chunk-GQNJO2P2.js +0 -102
- package/dist/chunk-GQNJO2P2.js.map +0 -7
- package/dist/chunk-IJGF2ETG.js +0 -68
- package/dist/chunk-IJGF2ETG.js.map +0 -7
- package/dist/chunk-ILYDELUE.js +0 -46
- package/dist/chunk-ILYDELUE.js.map +0 -7
- package/dist/chunk-J6QFW4LD.js +0 -55
- package/dist/chunk-J6QFW4LD.js.map +0 -7
- package/dist/chunk-J7N3DLH6.js +0 -117
- package/dist/chunk-J7N3DLH6.js.map +0 -7
- package/dist/chunk-JFDWKLQC.js +0 -23
- package/dist/chunk-JFDWKLQC.js.map +0 -7
- package/dist/chunk-JULOLTZS.js +0 -61
- package/dist/chunk-JULOLTZS.js.map +0 -7
- package/dist/chunk-KJKBDQ4D.js +0 -220
- package/dist/chunk-KJKBDQ4D.js.map +0 -7
- package/dist/chunk-KRBQLMOP.js +0 -41
- package/dist/chunk-KRBQLMOP.js.map +0 -7
- package/dist/chunk-LXKPNATC.js +0 -624
- package/dist/chunk-LXKPNATC.js.map +0 -7
- package/dist/chunk-NSBPE2FW.js +0 -17
- package/dist/chunk-NSBPE2FW.js.map +0 -7
- package/dist/chunk-OCKHW72U.js +0 -55
- package/dist/chunk-OCKHW72U.js.map +0 -7
- package/dist/chunk-OKWCKCPE.js +0 -414
- package/dist/chunk-OKWCKCPE.js.map +0 -7
- package/dist/chunk-OO7JDFS4.js +0 -157
- package/dist/chunk-OO7JDFS4.js.map +0 -7
- package/dist/chunk-OQBBVTDG.js.map +0 -7
- package/dist/chunk-OQSQC7VB.js +0 -77
- package/dist/chunk-OQSQC7VB.js.map +0 -7
- package/dist/chunk-PED7L5GX.js +0 -92
- package/dist/chunk-PED7L5GX.js.map +0 -7
- package/dist/chunk-QJQS6TUN.js +0 -44
- package/dist/chunk-QJQS6TUN.js.map +0 -7
- package/dist/chunk-QN34TGD3.js +0 -83
- package/dist/chunk-QN34TGD3.js.map +0 -7
- package/dist/chunk-QSN77T7C.js +0 -250
- package/dist/chunk-QSN77T7C.js.map +0 -7
- package/dist/chunk-R56PH7YE.js +0 -78
- package/dist/chunk-R56PH7YE.js.map +0 -7
- package/dist/chunk-RAXRZYOB.js +0 -299
- package/dist/chunk-RAXRZYOB.js.map +0 -7
- package/dist/chunk-RH7NHVSK.js +0 -16
- package/dist/chunk-RH7NHVSK.js.map +0 -7
- package/dist/chunk-RMTT4KDY.js +0 -191
- package/dist/chunk-RMTT4KDY.js.map +0 -7
- package/dist/chunk-RQ2TMLKG.js +0 -142
- package/dist/chunk-RQ2TMLKG.js.map +0 -7
- package/dist/chunk-SO3T35U7.js +0 -80
- package/dist/chunk-SO3T35U7.js.map +0 -7
- package/dist/chunk-UKMLICWS.js +0 -51
- package/dist/chunk-UKMLICWS.js.map +0 -7
- package/dist/chunk-VA5FFOS7.js +0 -243
- package/dist/chunk-VA5FFOS7.js.map +0 -7
- package/dist/chunk-W2YERO7E.js +0 -170
- package/dist/chunk-W2YERO7E.js.map +0 -7
- package/dist/chunk-W53GKIZQ.js +0 -50
- package/dist/chunk-W53GKIZQ.js.map +0 -7
- package/dist/chunk-WA4MWYRW.js +0 -254
- package/dist/chunk-WA4MWYRW.js.map +0 -7
- package/dist/chunk-XC6M4CSW.js +0 -44
- package/dist/chunk-XC6M4CSW.js.map +0 -7
- package/dist/chunk-XK42ZBDE.js +0 -23
- package/dist/chunk-XK42ZBDE.js.map +0 -7
- package/dist/chunk-XUV6DDBF.js +0 -124
- package/dist/chunk-XUV6DDBF.js.map +0 -7
- package/dist/chunk-XZXS2W24.js.map +0 -7
- package/dist/chunk-Y3WD5CSW.js +0 -49
- package/dist/chunk-Y3WD5CSW.js.map +0 -7
- package/dist/chunk-Y4BJXXYV.js +0 -148
- package/dist/chunk-Y4BJXXYV.js.map +0 -7
- package/dist/chunk-YNIH3KMU.js +0 -121
- package/dist/chunk-YNIH3KMU.js.map +0 -7
- package/dist/chunk-ZUJBM6HE.js +0 -125
- package/dist/chunk-ZUJBM6HE.js.map +0 -7
- package/dist/commands/antigravity/backfill.js.map +0 -7
- package/dist/commands/auth/create-key.js.map +0 -7
- package/dist/commands/auth/login.js.map +0 -7
- package/dist/commands/auth/logout.js.map +0 -7
- package/dist/commands/auth/revoke-key.js.map +0 -7
- package/dist/commands/auth/signup.js.map +0 -7
- package/dist/commands/auth/status.js.map +0 -7
- package/dist/commands/catalog/index.js.map +0 -7
- package/dist/commands/catalog/recommend.js.map +0 -7
- package/dist/commands/convert.js.map +0 -7
- package/dist/commands/converters.js.map +0 -7
- package/dist/commands/crons/list.js.map +0 -7
- package/dist/commands/crons/run.js.map +0 -7
- package/dist/commands/dashboard.js.map +0 -7
- package/dist/commands/diagnostics.js.map +0 -7
- package/dist/commands/eval-targets/create.js.map +0 -7
- package/dist/commands/eval-targets/list.js.map +0 -7
- package/dist/commands/evals/agent.js.map +0 -7
- package/dist/commands/evals/benchmark.js.map +0 -7
- package/dist/commands/evals/compare.js.map +0 -7
- package/dist/commands/evals/create.js.map +0 -7
- package/dist/commands/evals/optimize-description.js.map +0 -7
- package/dist/commands/evals/optimize.js.map +0 -7
- package/dist/commands/evals/run.js.map +0 -7
- package/dist/commands/gateway/ensure.js.map +0 -7
- package/dist/commands/hooks/emit-instructions.js.map +0 -7
- package/dist/commands/hooks/install.js.map +0 -7
- package/dist/commands/hooks/run.js.map +0 -7
- package/dist/commands/hooks/uninstall.js.map +0 -7
- package/dist/commands/import.js.map +0 -7
- package/dist/commands/instructions/create.js.map +0 -7
- package/dist/commands/instructions/delete.js.map +0 -7
- package/dist/commands/instructions/fork.js.map +0 -7
- package/dist/commands/instructions/get.js.map +0 -7
- package/dist/commands/instructions/list.js.map +0 -7
- package/dist/commands/instructions/promote.js.map +0 -7
- package/dist/commands/instructions/pull.js.map +0 -7
- package/dist/commands/instructions/upstream.js.map +0 -7
- package/dist/commands/instructions/versions.js.map +0 -7
- package/dist/commands/launcher.js.map +0 -7
- package/dist/commands/org/current.js.map +0 -7
- package/dist/commands/org/list.js.map +0 -7
- package/dist/commands/org/switch.js.map +0 -7
- package/dist/commands/packs/add-instruction.js.map +0 -7
- package/dist/commands/packs/apply.js.map +0 -7
- package/dist/commands/packs/create.js.map +0 -7
- package/dist/commands/packs/fork.js.map +0 -7
- package/dist/commands/packs/get.js.map +0 -7
- package/dist/commands/packs/list.js.map +0 -7
- package/dist/commands/packs/remove-instruction.js.map +0 -7
- package/dist/commands/proxy/env.js.map +0 -7
- package/dist/commands/proxy/logs.js.map +0 -7
- package/dist/commands/proxy/restart.js.map +0 -7
- package/dist/commands/proxy/setup.js.map +0 -7
- package/dist/commands/proxy/start.js.map +0 -7
- package/dist/commands/proxy/status.js.map +0 -7
- package/dist/commands/proxy/stop.js.map +0 -7
- package/dist/commands/research/backfill.js.map +0 -7
- package/dist/commands/research/tail-transcript.js.map +0 -7
- package/dist/commands/service/install.js.map +0 -7
- package/dist/commands/service/status.js.map +0 -7
- package/dist/commands/service/uninstall.js.map +0 -7
- package/dist/commands/sessions/analyze.js.map +0 -7
- package/dist/commands/sessions/end.js.map +0 -7
- package/dist/commands/sessions/import.js.map +0 -7
- package/dist/commands/sessions/list.js.map +0 -7
- package/dist/commands/sessions/start.js.map +0 -7
- package/dist/commands/setup.js.map +0 -7
- package/dist/commands/skills/export-approved.js.map +0 -7
- package/dist/commands/skills/install.js.map +0 -7
- package/dist/commands/skills/list.js.map +0 -7
- package/dist/commands/skills/publish.js.map +0 -7
- package/dist/commands/skills/search.js.map +0 -7
- package/dist/commands/skills/update.js.map +0 -7
- package/dist/commands/suggestions/accept.js.map +0 -7
- package/dist/commands/suggestions/dismiss.js.map +0 -7
- package/dist/commands/suggestions/list.js.map +0 -7
- package/dist/commands/suggestions/refresh.js.map +0 -7
- package/dist/cursor-BGQJ4CWI.js +0 -134
- package/dist/cursor-BGQJ4CWI.js.map +0 -7
- package/dist/dashboard/Dashboard.js.map +0 -7
- package/dist/dashboard/data.js.map +0 -7
- package/dist/dashboard/launcher/ActiveJobsList.js.map +0 -7
- package/dist/dashboard/launcher/LauncherOverlay.js.map +0 -7
- package/dist/dashboard/launcher/NewLaunchForm.js.map +0 -7
- package/dist/dashboard/launcher/api.js.map +0 -7
- package/dist/dashboard/launcher/tool-labels.js.map +0 -7
- package/dist/dashboard/launcher/types.js.map +0 -7
- package/dist/dashboard/launcher/useLaunchJobs.js.map +0 -7
- package/dist/dashboard/panels/friction.js.map +0 -7
- package/dist/dashboard/panels/insights.js.map +0 -7
- package/dist/dashboard/panels/instructions.js.map +0 -7
- package/dist/dashboard/panels/outcomes.js.map +0 -7
- package/dist/dashboard/panels/overview.js.map +0 -7
- package/dist/dashboard/panels/shared.js.map +0 -7
- package/dist/dashboard/types.js.map +0 -7
- package/dist/dist-UNYP5PJR.js +0 -56
- package/dist/dist-UNYP5PJR.js.map +0 -7
- package/dist/index.js.map +0 -7
- package/dist/lib/active-org-refresh.js.map +0 -7
- package/dist/lib/active-org.js.map +0 -7
- package/dist/lib/agent-loop.js.map +0 -7
- package/dist/lib/antigravity-transcript.js.map +0 -7
- package/dist/lib/api-client.js.map +0 -7
- package/dist/lib/auth.js.map +0 -7
- package/dist/lib/capture-hosts.js.map +0 -7
- package/dist/lib/detect-languages.js.map +0 -7
- package/dist/lib/detect-tmux.js.map +0 -7
- package/dist/lib/diagnostics.js.map +0 -7
- package/dist/lib/editor-proxy-pins.js.map +0 -7
- package/dist/lib/ensure-api-key.js.map +0 -7
- package/dist/lib/eval-analyzer.js.map +0 -7
- package/dist/lib/eval-benchmark.js.map +0 -7
- package/dist/lib/eval-conversation-handler.js.map +0 -7
- package/dist/lib/eval-executor.js.map +0 -7
- package/dist/lib/eval-grader.js.map +0 -7
- package/dist/lib/eval-meta-judge.js.map +0 -7
- package/dist/lib/eval-trigger-tester.js.map +0 -7
- package/dist/lib/eval-worktree.js.map +0 -7
- package/dist/lib/gateway-entry.js.map +0 -7
- package/dist/lib/gateway-identity.js.map +0 -7
- package/dist/lib/gateway-lifecycle.js.map +0 -7
- package/dist/lib/gateway.js.map +0 -7
- package/dist/lib/handlers/cron-refresh-skills.js.map +0 -7
- package/dist/lib/handlers/cron-suggest-instructions.js.map +0 -7
- package/dist/lib/handlers/process-announcement.js.map +0 -7
- package/dist/lib/handlers/process-changelog.js.map +0 -7
- package/dist/lib/handlers/process-conversation.js.map +0 -7
- package/dist/lib/handlers/process-eval.js.map +0 -7
- package/dist/lib/handlers/process-insights.js.map +0 -7
- package/dist/lib/handlers/process-launch.js.map +0 -7
- package/dist/lib/handlers/process-run-cleanup.js.map +0 -7
- package/dist/lib/handlers/process-send-message.js.map +0 -7
- package/dist/lib/handlers/process-session-goal.js.map +0 -7
- package/dist/lib/hooks-config.js.map +0 -7
- package/dist/lib/insights-prompts.js.map +0 -7
- package/dist/lib/instruction-snapshot.js.map +0 -7
- package/dist/lib/job-helpers.js.map +0 -7
- package/dist/lib/llm-client.js.map +0 -7
- package/dist/lib/manual-tasks-meta.js.map +0 -7
- package/dist/lib/manual-tasks.js.map +0 -7
- package/dist/lib/output.js.map +0 -7
- package/dist/lib/permanent-job-error.js.map +0 -7
- package/dist/lib/pid-identity.js.map +0 -7
- package/dist/lib/proxy-respawn.js.map +0 -7
- package/dist/lib/research-client.js.map +0 -7
- package/dist/lib/setup-steps.js.map +0 -7
- package/dist/lib/skills-loader.js.map +0 -7
- package/dist/lib/statusline-shim.js.map +0 -7
- package/dist/lib/telemetry.js.map +0 -7
- package/dist/lib/terminal-target-heartbeat.js.map +0 -7
- package/dist/lib/tmux.js.map +0 -7
- package/dist/lib/transcript-rate-limit-watcher.js.map +0 -7
- package/dist/lib/types.js.map +0 -7
- package/dist/lib/version-check.js.map +0 -7
- package/dist/lib/which.js.map +0 -7
- package/dist/lib/worktree.js.map +0 -7
- package/dist/opencode-IQ7X5YMJ.js +0 -190
- package/dist/opencode-IQ7X5YMJ.js.map +0 -7
|
@@ -1,169 +1,28 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
}
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
}
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
import "../../chunk-EKJ2CABV.js";
|
|
15
|
-
import {
|
|
16
|
-
apiGet,
|
|
17
|
-
apiPost,
|
|
18
|
-
apiPut
|
|
19
|
-
} from "../../chunk-ZUJBM6HE.js";
|
|
20
|
-
import "../../chunk-W2YERO7E.js";
|
|
21
|
-
import "../../chunk-NSBPE2FW.js";
|
|
22
|
-
|
|
23
|
-
// src/commands/evals/optimize-description.ts
|
|
24
|
-
import { Args, Flags } from "@oclif/core";
|
|
25
|
-
var EvalOptimizeDescriptionCommand = class _EvalOptimizeDescriptionCommand extends BaseCommand {
|
|
26
|
-
static description = "Optimize eval target description for better triggering accuracy";
|
|
27
|
-
static examples = [
|
|
28
|
-
"<%= config.bin %> evals optimize-description <target-id>",
|
|
29
|
-
"<%= config.bin %> evals optimize-description <target-id> --queries 10"
|
|
30
|
-
];
|
|
31
|
-
static args = {
|
|
32
|
-
targetId: Args.string({ description: "Eval target ID", required: true })
|
|
33
|
-
};
|
|
34
|
-
static flags = {
|
|
35
|
-
"max-iterations": Flags.integer({ description: "Maximum optimization iterations", default: 5 }),
|
|
36
|
-
model: Flags.string({ description: "Model for trigger testing and improvement" }),
|
|
37
|
-
queries: Flags.integer({ description: "Number of trigger queries to generate", default: 20 }),
|
|
38
|
-
"runs-per-query": Flags.integer({ description: "Runs per query for reliability", default: 3 }),
|
|
39
|
-
"poll-interval": Flags.integer({ description: "Ms between approval polls", default: 5e3 }),
|
|
40
|
-
timeout: Flags.integer({ description: "Timeout per eval in ms", default: 12e4 })
|
|
41
|
-
};
|
|
42
|
-
async run() {
|
|
43
|
-
const { args, flags } = await this.parse(_EvalOptimizeDescriptionCommand);
|
|
44
|
-
this.requireAuth();
|
|
45
|
-
const target = await apiGet(`/api/eval-targets/${args.targetId}`);
|
|
46
|
-
this.log(`Optimizing description for "${target.name}" v${target.version}`);
|
|
47
|
-
this.log(`Current description: "${target.description}"`);
|
|
48
|
-
this.log("");
|
|
49
|
-
this.log("Generating trigger queries...");
|
|
50
|
-
const queries = await this.generateTriggerQueries(target, flags.queries, flags);
|
|
51
|
-
const triggerEval = await apiPost(`/api/eval-targets/${args.targetId}/trigger-evals`, {
|
|
52
|
-
queries
|
|
53
|
-
});
|
|
54
|
-
this.log(`
|
|
55
|
-
Generated ${queries.length} trigger queries.`);
|
|
56
|
-
this.log(`Review and approve them at:`);
|
|
57
|
-
this.log(` http://localhost:5174/eval-targets/${args.targetId}`);
|
|
58
|
-
this.log(` (Triggers tab \u2192 click to review)`);
|
|
59
|
-
this.log("\nWaiting for approval...");
|
|
60
|
-
let approved = false;
|
|
61
|
-
while (!approved) {
|
|
62
|
-
await new Promise((resolve) => setTimeout(resolve, flags["poll-interval"]));
|
|
63
|
-
const sets = await apiGet(
|
|
64
|
-
`/api/eval-targets/${args.targetId}/trigger-evals?status=approved`
|
|
65
|
-
);
|
|
66
|
-
const current = sets.find((s) => s.id === triggerEval.id);
|
|
67
|
-
if (current) {
|
|
68
|
-
approved = true;
|
|
69
|
-
queries.length = 0;
|
|
70
|
-
queries.push(...current.queries);
|
|
71
|
-
this.log("Trigger queries approved!");
|
|
72
|
-
}
|
|
73
|
-
}
|
|
74
|
-
const indices = queries.map((_, i) => i).sort(() => Math.random() - 0.5);
|
|
75
|
-
const splitIdx = Math.ceil(indices.length * 0.6);
|
|
76
|
-
const trainIndices = indices.slice(0, splitIdx);
|
|
77
|
-
const testIndices = indices.slice(splitIdx);
|
|
78
|
-
await apiPut(`/api/trigger-evals/${triggerEval.id}`, {
|
|
79
|
-
trainIndices,
|
|
80
|
-
testIndices
|
|
81
|
-
});
|
|
82
|
-
const trainQueries = trainIndices.map((i) => queries[i]);
|
|
83
|
-
const testQueries = testIndices.map((i) => queries[i]);
|
|
84
|
-
this.log(`
|
|
85
|
-
Split: ${trainQueries.length} train, ${testQueries.length} test`);
|
|
86
|
-
let currentDescription = target.description;
|
|
87
|
-
let bestDescription = currentDescription;
|
|
88
|
-
let bestTestScore = 0;
|
|
89
|
-
for (let iteration = 1; iteration <= flags["max-iterations"]; iteration++) {
|
|
90
|
-
this.log(`
|
|
91
|
-
=== Iteration ${iteration} ===`);
|
|
92
|
-
this.log(`Description: "${currentDescription}"`);
|
|
93
|
-
this.log("Testing train set...");
|
|
94
|
-
const trainResults = await this.testQuerySet(trainQueries, target, currentDescription, flags);
|
|
95
|
-
const trainScore = computeTriggerScore(trainResults);
|
|
96
|
-
this.log(`Train accuracy: ${(trainScore * 100).toFixed(1)}%`);
|
|
97
|
-
this.printTriggerResults(trainResults);
|
|
98
|
-
this.log("Testing test set...");
|
|
99
|
-
const testResults = await this.testQuerySet(testQueries, target, currentDescription, flags);
|
|
100
|
-
const testScore = computeTriggerScore(testResults);
|
|
101
|
-
this.log(`Test accuracy: ${(testScore * 100).toFixed(1)}%`);
|
|
102
|
-
await apiPost("/api/optimization-runs", {
|
|
103
|
-
evalTargetId: target.id,
|
|
104
|
-
type: "description",
|
|
105
|
-
iteration,
|
|
106
|
-
inputVersion: target.version,
|
|
107
|
-
trainScore,
|
|
108
|
-
testScore,
|
|
109
|
-
analysis: {
|
|
110
|
-
description: currentDescription,
|
|
111
|
-
trainResults: trainResults.map((r) => ({ query: r.query, shouldTrigger: r.shouldTrigger, didTrigger: r.didTrigger, correct: r.correct })),
|
|
112
|
-
testResults: testResults.map((r) => ({ query: r.query, shouldTrigger: r.shouldTrigger, didTrigger: r.didTrigger, correct: r.correct }))
|
|
113
|
-
},
|
|
114
|
-
status: "completed"
|
|
115
|
-
});
|
|
116
|
-
if (testScore > bestTestScore) {
|
|
117
|
-
bestTestScore = testScore;
|
|
118
|
-
bestDescription = currentDescription;
|
|
119
|
-
}
|
|
120
|
-
if (trainScore >= 1) {
|
|
121
|
-
this.log("Perfect train score. Stopping.");
|
|
122
|
-
break;
|
|
123
|
-
}
|
|
124
|
-
this.log("Generating improved description...");
|
|
125
|
-
const failures = trainResults.filter((r) => !r.correct);
|
|
126
|
-
currentDescription = await this.improveDescription(target, currentDescription, failures, flags);
|
|
127
|
-
}
|
|
128
|
-
this.log(`
|
|
129
|
-
=== Optimization Complete ===`);
|
|
130
|
-
this.log(`Best test accuracy: ${(bestTestScore * 100).toFixed(1)}%`);
|
|
131
|
-
this.log(`
|
|
132
|
-
Before: "${target.description}"`);
|
|
133
|
-
this.log(`After: "${bestDescription}"`);
|
|
134
|
-
if (bestDescription !== target.description) {
|
|
135
|
-
await apiPut(`/api/eval-targets/${target.id}`, {
|
|
136
|
-
description: bestDescription
|
|
137
|
-
});
|
|
138
|
-
this.log("\nDescription updated.");
|
|
139
|
-
} else {
|
|
140
|
-
this.log("\nNo improvement found. Description unchanged.");
|
|
141
|
-
}
|
|
142
|
-
}
|
|
143
|
-
async generateTriggerQueries(target, count, flags) {
|
|
144
|
-
let sessionSeed = "";
|
|
145
|
-
try {
|
|
146
|
-
const sessions = await apiGet("/api/sessions?limit=10");
|
|
147
|
-
if (sessions.length > 0) {
|
|
148
|
-
sessionSeed = "\n\nReal user prompts from recent sessions (use as inspiration for realistic queries):\n" + sessions.slice(0, 5).map((s) => `- Session ${s.id}`).join("\n");
|
|
149
|
-
}
|
|
150
|
-
} catch {
|
|
151
|
-
}
|
|
152
|
-
const shouldTriggerCount = Math.ceil(count / 2);
|
|
153
|
-
const shouldNotTriggerCount = count - shouldTriggerCount;
|
|
154
|
-
const prompt = `Generate ${count} trigger eval queries for a skill/instruction.
|
|
1
|
+
import{a as S,b as P,c as R}from"../../chunk-KV325CXF.js";import{a as b}from"../../chunk-MGHFDHHM.js";import{a as O}from"../../chunk-PPKPCXJD.js";import"../../chunk-IZTIYR5Q.js";import"../../chunk-V44SRRYR.js";import"../../chunk-U6XEU573.js";import"../../chunk-VLOX6KOM.js";import"../../chunk-3AV3GTMP.js";import{e as y,f as q,g as w}from"../../chunk-H2ORWIEY.js";import"../../chunk-O6C2UGBY.js";import{a as N}from"../../chunk-ZLLER3HP.js";import{Args as C,Flags as T}from"@oclif/core";var A=process.env.RULEMETRIC_WEB_URL??"https://rulemetric.com",k=30*6e4,I=class M extends O{static{N(this,"EvalOptimizeDescriptionCommand")}static description="Optimize eval target description for better triggering accuracy";static examples=["<%= config.bin %> evals optimize-description <target-id>","<%= config.bin %> evals optimize-description <target-id> --queries 10"];static args={targetId:C.string({description:"Eval target ID",required:!0})};static flags={"max-iterations":T.integer({description:"Maximum optimization iterations",default:5}),model:T.string({description:"Model for trigger testing and improvement"}),queries:T.integer({description:"Number of trigger queries to generate",default:20}),"runs-per-query":T.integer({description:"Runs per query for reliability",default:3}),"poll-interval":T.integer({description:"Ms between approval polls",default:5e3}),timeout:T.integer({description:"Timeout per eval in ms",default:12e4})};async run(){let{args:i,flags:t}=await this.parse(M);this.requireAuth();let e=await y(`/api/eval-targets/${i.targetId}`);this.log(`Optimizing description for "${e.name}" v${e.version}`),this.log(`Current description: "${e.description}"`),this.log(S),this.log(""),this.log("Generating trigger queries...");let r=await this.generateTriggerQueries(e,t.queries,t),a=await q(`/api/eval-targets/${i.targetId}/trigger-evals`,{queries:r}),g=`${A}/eval-targets/${i.targetId}`;this.log(`
|
|
2
|
+
Generated ${r.length} trigger queries.`),this.log("Review and approve them at:"),this.log(` ${g}`),this.log(" (Triggers tab \u2192 click to review)"),this.log(`
|
|
3
|
+
Waiting for approval...`);let u=Date.now(),d=!1;for(;!d;){Date.now()-u>k&&this.error(`Timed out after ${k/6e4} minutes waiting for approval. Approve the trigger set at ${g} (Triggers tab) and re-run this command.`),await new Promise(h=>setTimeout(h,t["poll-interval"]));let p=(await y(`/api/eval-targets/${i.targetId}/trigger-evals?status=approved`)).find(h=>h.id===a.id);p&&(d=!0,r.length=0,r.push(...p.queries),this.log("Trigger queries approved!"))}let n=r.map((c,p)=>p).sort(()=>Math.random()-.5),o=Math.ceil(n.length*.6),s=n.slice(0,o),x=n.slice(o);await w(`/api/trigger-evals/${a.id}`,{trainIndices:s,testIndices:x});let D=s.map(c=>r[c]),_=x.map(c=>r[c]);this.log(`
|
|
4
|
+
Split: ${D.length} train, ${_.length} test`);let m=e.description,v=m,$=0;for(let c=1;c<=t["max-iterations"];c++){this.log(`
|
|
5
|
+
=== Iteration ${c} ===`),this.log(`Description: "${m}"`),this.log("Testing train set...");let p=await this.testQuerySet(D,e,m,t),h=R(p);this.log(`Train accuracy: ${(h*100).toFixed(1)}%`),this.printTriggerResults(p),this.log("Testing test set...");let E=await this.testQuerySet(_,e,m,t),f=R(E);if(this.log(`Test accuracy: ${(f*100).toFixed(1)}%`),await q("/api/optimization-runs",{evalTargetId:e.id,type:"description",iteration:c,inputVersion:e.version,trainScore:h,testScore:f,analysis:{description:m,trainResults:p.map(l=>({query:l.query,shouldTrigger:l.shouldTrigger,didTrigger:l.didTrigger,correct:l.correct})),testResults:E.map(l=>({query:l.query,shouldTrigger:l.shouldTrigger,didTrigger:l.didTrigger,correct:l.correct}))},status:"completed"}),f>$&&($=f,v=m),h>=1){this.log("Perfect train score. Stopping.");break}this.log("Generating improved description...");let Q=p.filter(l=>!l.correct);m=await this.improveDescription(e,m,Q,t)}this.log(`
|
|
6
|
+
=== Optimization Complete ===`),this.log(`Best test accuracy: ${($*100).toFixed(1)}%`),this.log(S),this.log(`
|
|
7
|
+
Before: "${e.description}"`),this.log(`After: "${v}"`),v!==e.description?(await w(`/api/eval-targets/${e.id}`,{description:v}),this.log(`
|
|
8
|
+
Description updated.`)):this.log(`
|
|
9
|
+
No improvement found. Description unchanged.`)}async generateTriggerQueries(i,t,e){let r="";try{let o=(await y("/api/sessions?limit=10")).map(s=>s.firstUserPrompt??s.lastMessagePreview??"").filter(s=>s.trim().length>0).slice(0,5);o.length>0&&(r=`
|
|
10
|
+
|
|
11
|
+
Real user prompts from recent sessions (use as inspiration for realistic queries):
|
|
12
|
+
`+o.map(s=>`- ${s}`).join(`
|
|
13
|
+
`))}catch{}let a=Math.ceil(t/2),g=t-a,u=`Generate ${t} trigger eval queries for a skill/instruction.
|
|
155
14
|
|
|
156
15
|
## Skill Name
|
|
157
|
-
${
|
|
16
|
+
${i.name}
|
|
158
17
|
|
|
159
18
|
## Skill Description
|
|
160
|
-
${
|
|
19
|
+
${i.description}
|
|
161
20
|
|
|
162
21
|
## Skill Content (first 1000 chars)
|
|
163
|
-
${
|
|
164
|
-
${
|
|
22
|
+
${i.content.slice(0,1e3)}
|
|
23
|
+
${r}
|
|
165
24
|
|
|
166
|
-
Generate ${
|
|
25
|
+
Generate ${a} should-trigger queries and ${g} should-not-trigger queries.
|
|
167
26
|
|
|
168
27
|
Rules:
|
|
169
28
|
- Queries must be realistic \u2014 what a real user would actually type
|
|
@@ -176,81 +35,20 @@ Respond with ONLY a valid JSON array:
|
|
|
176
35
|
[
|
|
177
36
|
{"query": "realistic user prompt...", "should_trigger": true},
|
|
178
37
|
{"query": "another prompt...", "should_trigger": false}
|
|
179
|
-
]
|
|
180
|
-
|
|
181
|
-
prompt,
|
|
182
|
-
timeout: flags.timeout,
|
|
183
|
-
model: flags.model
|
|
184
|
-
});
|
|
185
|
-
try {
|
|
186
|
-
let jsonStr = result.output.trim();
|
|
187
|
-
const jsonMatch = jsonStr.match(/```(?:json)?\s*([\s\S]*?)```/);
|
|
188
|
-
if (jsonMatch) jsonStr = jsonMatch[1].trim();
|
|
189
|
-
const arrMatch = jsonStr.match(/\[[\s\S]*\]/);
|
|
190
|
-
if (arrMatch) jsonStr = arrMatch[0];
|
|
191
|
-
return JSON.parse(jsonStr);
|
|
192
|
-
} catch {
|
|
193
|
-
return [
|
|
194
|
-
{ query: `Help me use ${target.name}`, should_trigger: true },
|
|
195
|
-
{ query: "Write a hello world program", should_trigger: false }
|
|
196
|
-
];
|
|
197
|
-
}
|
|
198
|
-
}
|
|
199
|
-
async testQuerySet(queries, target, description, flags) {
|
|
200
|
-
const results = [];
|
|
201
|
-
for (const q of queries) {
|
|
202
|
-
const runs = [];
|
|
203
|
-
for (let run = 0; run < flags["runs-per-query"]; run++) {
|
|
204
|
-
const result = await testTrigger({
|
|
205
|
-
query: q.query,
|
|
206
|
-
shouldTrigger: q.should_trigger,
|
|
207
|
-
skillName: target.name,
|
|
208
|
-
skillDescription: description,
|
|
209
|
-
skillContent: target.content,
|
|
210
|
-
model: flags.model,
|
|
211
|
-
timeout: flags.timeout
|
|
212
|
-
});
|
|
213
|
-
runs.push(result);
|
|
214
|
-
}
|
|
215
|
-
const triggerCount = runs.filter((r) => r.didTrigger).length;
|
|
216
|
-
const didTrigger = triggerCount > flags["runs-per-query"] / 2;
|
|
217
|
-
results.push({
|
|
218
|
-
query: q.query,
|
|
219
|
-
shouldTrigger: q.should_trigger,
|
|
220
|
-
didTrigger,
|
|
221
|
-
correct: didTrigger === q.should_trigger,
|
|
222
|
-
confidence: runs.reduce((sum, r) => sum + r.confidence, 0) / runs.length,
|
|
223
|
-
evidence: runs[0].evidence
|
|
224
|
-
});
|
|
225
|
-
}
|
|
226
|
-
return results;
|
|
227
|
-
}
|
|
228
|
-
printTriggerResults(results) {
|
|
229
|
-
for (const r of results) {
|
|
230
|
-
const status = r.correct ? "OK" : "FAIL";
|
|
231
|
-
const expected = r.shouldTrigger ? "should trigger" : "should NOT trigger";
|
|
232
|
-
const actual = r.didTrigger ? "triggered" : "did NOT trigger";
|
|
233
|
-
this.log(` [${status}] ${expected} / ${actual}: "${r.query.slice(0, 60)}..."`);
|
|
234
|
-
}
|
|
235
|
-
}
|
|
236
|
-
async improveDescription(target, currentDescription, failures, flags) {
|
|
237
|
-
const failureText = failures.map((f) => {
|
|
238
|
-
const expected = f.shouldTrigger ? "SHOULD trigger but DID NOT" : "SHOULD NOT trigger but DID";
|
|
239
|
-
return `- ${expected}: "${f.query}"`;
|
|
240
|
-
}).join("\n");
|
|
241
|
-
const prompt = `You are optimizing a skill description for better triggering accuracy.
|
|
38
|
+
]`,d=await b({prompt:u,timeout:e.timeout,model:e.model});try{let n=d.output.trim(),o=n.match(/```(?:json)?\s*([\s\S]*?)```/);o&&(n=o[1].trim());let s=n.match(/\[[\s\S]*\]/);return s&&(n=s[0]),JSON.parse(n)}catch{return[{query:`Help me use ${i.name}`,should_trigger:!0},{query:"Write a hello world program",should_trigger:!1}]}}async testQuerySet(i,t,e,r){let a=[];for(let g of i){let u=[];for(let o=0;o<r["runs-per-query"];o++){let s=await P({query:g.query,shouldTrigger:g.should_trigger,skillName:t.name,skillDescription:e,skillContent:t.content,model:r.model,timeout:r.timeout});u.push(s)}let n=u.filter(o=>o.didTrigger).length>r["runs-per-query"]/2;a.push({query:g.query,shouldTrigger:g.should_trigger,didTrigger:n,correct:n===g.should_trigger,confidence:u.reduce((o,s)=>o+s.confidence,0)/u.length,evidence:u[0].evidence})}return a}printTriggerResults(i){for(let t of i){let e=t.correct?"OK":"FAIL",r=t.shouldTrigger?"should trigger":"should NOT trigger",a=t.didTrigger?"triggered":"did NOT trigger";this.log(` [${e}] ${r} / ${a}: "${t.query.slice(0,60)}..."`)}}async improveDescription(i,t,e,r){let a=e.map(d=>`- ${d.shouldTrigger?"SHOULD trigger but DID NOT":"SHOULD NOT trigger but DID"}: "${d.query}"`).join(`
|
|
39
|
+
`),g=`You are optimizing a skill description for better triggering accuracy.
|
|
242
40
|
|
|
243
41
|
## Skill Name
|
|
244
|
-
${
|
|
42
|
+
${i.name}
|
|
245
43
|
|
|
246
44
|
## Current Description
|
|
247
|
-
${
|
|
45
|
+
${t}
|
|
248
46
|
|
|
249
47
|
## Skill Content (first 500 chars)
|
|
250
|
-
${
|
|
48
|
+
${i.content.slice(0,500)}
|
|
251
49
|
|
|
252
50
|
## Failures
|
|
253
|
-
${
|
|
51
|
+
${a}
|
|
254
52
|
|
|
255
53
|
Propose an improved description that:
|
|
256
54
|
- Fixes the trigger failures above
|
|
@@ -258,16 +56,4 @@ Propose an improved description that:
|
|
|
258
56
|
- Uses specific keywords and contexts for when to trigger
|
|
259
57
|
- Includes "Use PROACTIVELY" phrasing for important trigger contexts
|
|
260
58
|
|
|
261
|
-
Respond with ONLY the new description text (no JSON, no explanation, just the description string).`;
|
|
262
|
-
const result = await executeEval({
|
|
263
|
-
prompt,
|
|
264
|
-
timeout: flags.timeout,
|
|
265
|
-
model: flags.model
|
|
266
|
-
});
|
|
267
|
-
return result.output.trim().replace(/^["']|["']$/g, "");
|
|
268
|
-
}
|
|
269
|
-
};
|
|
270
|
-
export {
|
|
271
|
-
EvalOptimizeDescriptionCommand as default
|
|
272
|
-
};
|
|
273
|
-
//# sourceMappingURL=optimize-description.js.map
|
|
59
|
+
Respond with ONLY the new description text (no JSON, no explanation, just the description string).`;return(await b({prompt:g,timeout:r.timeout,model:r.model})).output.trim().replace(/^["']|["']$/g,"")}};export{I as default};
|
|
@@ -1,211 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
}
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
} from "../../chunk-YNIH3KMU.js";
|
|
7
|
-
import {
|
|
8
|
-
executeEval
|
|
9
|
-
} from "../../chunk-QN34TGD3.js";
|
|
10
|
-
import {
|
|
11
|
-
BaseCommand
|
|
12
|
-
} from "../../chunk-2HK7RGZB.js";
|
|
13
|
-
import "../../chunk-R56PH7YE.js";
|
|
14
|
-
import "../../chunk-GQNJO2P2.js";
|
|
15
|
-
import "../../chunk-37LN5O7M.js";
|
|
16
|
-
import "../../chunk-EKJ2CABV.js";
|
|
17
|
-
import {
|
|
18
|
-
apiGet,
|
|
19
|
-
apiPost,
|
|
20
|
-
apiPut
|
|
21
|
-
} from "../../chunk-ZUJBM6HE.js";
|
|
22
|
-
import "../../chunk-W2YERO7E.js";
|
|
23
|
-
import "../../chunk-NSBPE2FW.js";
|
|
24
|
-
|
|
25
|
-
// src/commands/evals/optimize.ts
|
|
26
|
-
import { Args, Flags } from "@oclif/core";
|
|
27
|
-
var EvalOptimizeCommand = class _EvalOptimizeCommand extends BaseCommand {
|
|
28
|
-
static description = "Iteratively optimize eval target content using eval results as feedback";
|
|
29
|
-
static examples = [
|
|
30
|
-
"<%= config.bin %> evals optimize <target-id>",
|
|
31
|
-
"<%= config.bin %> evals optimize <target-id> --auto --max-iterations 3"
|
|
32
|
-
];
|
|
33
|
-
static args = {
|
|
34
|
-
targetId: Args.string({ description: "Eval target ID", required: true })
|
|
35
|
-
};
|
|
36
|
-
static flags = {
|
|
37
|
-
auto: Flags.boolean({ description: "Skip interactive review, apply suggestions automatically", default: false }),
|
|
38
|
-
"max-iterations": Flags.integer({ description: "Maximum optimization iterations", default: 5 }),
|
|
39
|
-
threshold: Flags.string({ description: "Stop if pass rate reaches this threshold (0-1)", default: "0.95" }),
|
|
40
|
-
model: Flags.string({ description: "Model for eval execution" }),
|
|
41
|
-
"analyzer-model": Flags.string({ description: "Model for analysis/suggestions" }),
|
|
42
|
-
"runs-per-eval": Flags.integer({ description: "Runs per eval per iteration", default: 1 }),
|
|
43
|
-
parallel: Flags.integer({ description: "Number of parallel eval executions", default: 1 }),
|
|
44
|
-
timeout: Flags.integer({ description: "Timeout per eval in ms", default: 12e4 })
|
|
45
|
-
};
|
|
46
|
-
async run() {
|
|
47
|
-
const { args, flags } = await this.parse(_EvalOptimizeCommand);
|
|
48
|
-
this.requireAuth();
|
|
49
|
-
let target = await apiGet(`/api/eval-targets/${args.targetId}`);
|
|
50
|
-
const allEvals = await apiGet(`/api/eval-targets/${args.targetId}/evals`);
|
|
51
|
-
if (allEvals.length === 0) {
|
|
52
|
-
this.error("No evals found. Create some first with `rulemetric evals create`.");
|
|
53
|
-
}
|
|
54
|
-
const shuffled = [...allEvals].sort(() => Math.random() - 0.5);
|
|
55
|
-
const splitIdx = Math.ceil(shuffled.length * 0.6);
|
|
56
|
-
const trainEvals = shuffled.slice(0, splitIdx);
|
|
57
|
-
const testEvals = shuffled.slice(splitIdx);
|
|
58
|
-
this.log(`Optimizing "${target.name}" v${target.version}`);
|
|
59
|
-
this.log(`Evals: ${trainEvals.length} train, ${testEvals.length} test`);
|
|
60
|
-
const threshold = parseFloat(flags.threshold);
|
|
61
|
-
this.log(`Config: max ${flags["max-iterations"]} iterations, threshold ${(threshold * 100).toFixed(0)}%`);
|
|
62
|
-
if (flags.auto) this.log("Mode: auto (no interactive review)");
|
|
63
|
-
this.log("");
|
|
64
|
-
let bestVersion = target.version;
|
|
65
|
-
let bestTestScore = 0;
|
|
66
|
-
let noImprovementCount = 0;
|
|
67
|
-
for (let iteration = 1; iteration <= flags["max-iterations"]; iteration++) {
|
|
68
|
-
this.log(`
|
|
69
|
-
=== Iteration ${iteration} ===`);
|
|
70
|
-
const optRun = await apiPost("/api/optimization-runs", {
|
|
71
|
-
evalTargetId: target.id,
|
|
72
|
-
type: "content",
|
|
73
|
-
iteration,
|
|
74
|
-
inputVersion: target.version,
|
|
75
|
-
status: "running"
|
|
76
|
-
});
|
|
77
|
-
this.log("Running train evals...");
|
|
78
|
-
const trainResults = await this.runEvalsOnTarget(target, trainEvals, flags);
|
|
79
|
-
const trainScore = this.computePassRate(trainResults);
|
|
80
|
-
this.log(`Train pass rate: ${(trainScore * 100).toFixed(1)}%`);
|
|
81
|
-
if (trainScore >= threshold) {
|
|
82
|
-
this.log(`Train score ${(trainScore * 100).toFixed(1)}% >= threshold ${(threshold * 100).toFixed(0)}%. Stopping.`);
|
|
83
|
-
const testResults2 = await this.runEvalsOnTarget(target, testEvals, flags);
|
|
84
|
-
const testScore2 = this.computePassRate(testResults2);
|
|
85
|
-
this.log(`Final test score: ${(testScore2 * 100).toFixed(1)}%`);
|
|
86
|
-
await apiPost("/api/optimization-runs", {
|
|
87
|
-
evalTargetId: target.id,
|
|
88
|
-
type: "content",
|
|
89
|
-
iteration,
|
|
90
|
-
inputVersion: target.version,
|
|
91
|
-
trainScore,
|
|
92
|
-
testScore: testScore2,
|
|
93
|
-
analysis: { reason: "threshold_reached" },
|
|
94
|
-
status: "completed"
|
|
95
|
-
});
|
|
96
|
-
break;
|
|
97
|
-
}
|
|
98
|
-
this.log("Analyzing results...");
|
|
99
|
-
const evalResults = trainResults.map((r) => ({
|
|
100
|
-
evalName: r.evalName,
|
|
101
|
-
prompt: r.prompt,
|
|
102
|
-
expectations: r.expectations,
|
|
103
|
-
grading: r.grading
|
|
104
|
-
}));
|
|
105
|
-
const analysis = await analyzeEvalResults({
|
|
106
|
-
content: target.content,
|
|
107
|
-
evalResults,
|
|
108
|
-
model: flags["analyzer-model"] ?? flags.model,
|
|
109
|
-
timeout: flags.timeout
|
|
110
|
-
});
|
|
111
|
-
this.log(`
|
|
112
|
-
Assessment: ${analysis.overallAssessment}`);
|
|
113
|
-
this.log(`Suggestions: ${analysis.suggestions.length}`);
|
|
114
|
-
for (const s of analysis.suggestions) {
|
|
115
|
-
this.log(` - [${s.section}] ${s.reasoning}`);
|
|
116
|
-
}
|
|
117
|
-
if (!flags.auto) {
|
|
118
|
-
this.log("\nProposed changes above. Applying in auto mode would proceed automatically.");
|
|
119
|
-
this.log("(Interactive approval not yet implemented \u2014 use --auto for now)");
|
|
120
|
-
}
|
|
121
|
-
if (analysis.proposedContent === target.content) {
|
|
122
|
-
this.log("No changes proposed. Stopping.");
|
|
123
|
-
await apiPost("/api/optimization-runs", {
|
|
124
|
-
evalTargetId: target.id,
|
|
125
|
-
type: "content",
|
|
126
|
-
iteration,
|
|
127
|
-
inputVersion: target.version,
|
|
128
|
-
trainScore,
|
|
129
|
-
analysis: { reason: "no_changes" },
|
|
130
|
-
status: "completed"
|
|
131
|
-
});
|
|
132
|
-
break;
|
|
133
|
-
}
|
|
134
|
-
const newTarget = await apiPut(`/api/eval-targets/${target.id}`, {
|
|
135
|
-
content: analysis.proposedContent
|
|
136
|
-
});
|
|
137
|
-
this.log(`Created v${newTarget.version}`);
|
|
138
|
-
this.log("Running test evals...");
|
|
139
|
-
const testResults = await this.runEvalsOnTarget(newTarget, testEvals, flags);
|
|
140
|
-
const testScore = this.computePassRate(testResults);
|
|
141
|
-
this.log(`Test pass rate: ${(testScore * 100).toFixed(1)}%`);
|
|
142
|
-
await apiPost("/api/optimization-runs", {
|
|
143
|
-
evalTargetId: newTarget.id,
|
|
144
|
-
type: "content",
|
|
145
|
-
iteration,
|
|
146
|
-
inputVersion: target.version,
|
|
147
|
-
outputVersion: newTarget.version,
|
|
148
|
-
trainScore,
|
|
149
|
-
testScore,
|
|
150
|
-
analysis: {
|
|
151
|
-
suggestions: analysis.suggestions,
|
|
152
|
-
overallAssessment: analysis.overallAssessment
|
|
153
|
-
},
|
|
154
|
-
status: "completed"
|
|
155
|
-
});
|
|
156
|
-
if (testScore > bestTestScore) {
|
|
157
|
-
bestTestScore = testScore;
|
|
158
|
-
bestVersion = newTarget.version;
|
|
159
|
-
noImprovementCount = 0;
|
|
160
|
-
} else {
|
|
161
|
-
noImprovementCount++;
|
|
162
|
-
if (noImprovementCount >= 2) {
|
|
163
|
-
this.log("No improvement for 2 consecutive iterations. Stopping.");
|
|
164
|
-
break;
|
|
165
|
-
}
|
|
166
|
-
}
|
|
167
|
-
target = newTarget;
|
|
168
|
-
}
|
|
169
|
-
this.log(`
|
|
170
|
-
=== Optimization Complete ===`);
|
|
171
|
-
this.log(`Best version: v${bestVersion} (test score: ${(bestTestScore * 100).toFixed(1)}%)`);
|
|
172
|
-
}
|
|
173
|
-
async runEvalsOnTarget(target, evalSet, flags) {
|
|
174
|
-
const results = [];
|
|
175
|
-
for (const eval_ of evalSet) {
|
|
176
|
-
const execResult = await executeEval({
|
|
177
|
-
prompt: eval_.prompt,
|
|
178
|
-
targetContent: target.content,
|
|
179
|
-
timeout: flags.timeout,
|
|
180
|
-
model: flags.model
|
|
181
|
-
});
|
|
182
|
-
if (execResult.error) {
|
|
183
|
-
results.push({ evalName: eval_.name, prompt: eval_.prompt, expectations: eval_.expectations, grading: {} });
|
|
184
|
-
continue;
|
|
185
|
-
}
|
|
186
|
-
if (eval_.expectations.length > 0) {
|
|
187
|
-
const grading = await gradeEvalOutput({
|
|
188
|
-
prompt: eval_.prompt,
|
|
189
|
-
output: execResult.output,
|
|
190
|
-
expectations: eval_.expectations,
|
|
191
|
-
model: flags.model,
|
|
192
|
-
timeout: flags.timeout
|
|
193
|
-
});
|
|
194
|
-
results.push({ evalName: eval_.name, prompt: eval_.prompt, expectations: eval_.expectations, grading });
|
|
195
|
-
} else {
|
|
196
|
-
results.push({ evalName: eval_.name, prompt: eval_.prompt, expectations: eval_.expectations, grading: {} });
|
|
197
|
-
}
|
|
198
|
-
}
|
|
199
|
-
return results;
|
|
200
|
-
}
|
|
201
|
-
computePassRate(results) {
|
|
202
|
-
const graded = results.filter((r) => Object.keys(r.grading).length > 0);
|
|
203
|
-
if (graded.length === 0) return 0;
|
|
204
|
-
const rates = graded.map((r) => r.grading.summary?.pass_rate ?? 0);
|
|
205
|
-
return rates.reduce((a, b) => a + b, 0) / rates.length;
|
|
206
|
-
}
|
|
207
|
-
};
|
|
208
|
-
export {
|
|
209
|
-
EvalOptimizeCommand as default
|
|
210
|
-
};
|
|
211
|
-
//# sourceMappingURL=optimize.js.map
|
|
1
|
+
import{a as A}from"../../chunk-DSO2U4Y6.js";import{a as k}from"../../chunk-CADKBP3D.js";import{a as z}from"../../chunk-MGHFDHHM.js";import{a as I}from"../../chunk-PPKPCXJD.js";import"../../chunk-IZTIYR5Q.js";import"../../chunk-V44SRRYR.js";import"../../chunk-U6XEU573.js";import"../../chunk-VLOX6KOM.js";import"../../chunk-3AV3GTMP.js";import{e as y,f as m,g as R}from"../../chunk-H2ORWIEY.js";import"../../chunk-O6C2UGBY.js";import{a as b}from"../../chunk-ZLLER3HP.js";import{createInterface as C}from"node:readline/promises";import{Args as P,Flags as p}from"@oclif/core";var w=class N extends I{static{b(this,"EvalOptimizeCommand")}static description="Iteratively optimize eval target content using eval results as feedback";static examples=["<%= config.bin %> evals optimize <target-id>","<%= config.bin %> evals optimize <target-id> --auto --max-iterations 3"];static args={targetId:P.string({description:"Eval target ID",required:!0})};static flags={auto:p.boolean({description:"Skip interactive review, apply suggestions automatically",default:!1}),"max-iterations":p.integer({description:"Maximum optimization iterations",default:5}),threshold:p.string({description:"Stop if pass rate reaches this threshold (0-1)",default:"0.95"}),model:p.string({description:"Model for eval execution"}),"analyzer-model":p.string({description:"Model for analysis/suggestions"}),timeout:p.integer({description:"Timeout per eval in ms",default:12e4})};async run(){let{args:a,flags:e}=await this.parse(N);this.requireAuth();let t=await y(`/api/eval-targets/${a.targetId}`),i=await y(`/api/eval-targets/${a.targetId}/evals`);i.length===0&&this.error("No evals found. Create some first with `rulemetric evals create`.");let s=[...i].sort(()=>Math.random()-.5),g=Math.ceil(s.length*.6),u=s.slice(0,g),v=s.slice(g);this.log(`Optimizing "${t.name}" v${t.version}`),this.log(`Evals: ${u.length} train, ${v.length} test`);let h=parseFloat(e.threshold);this.log(`Config: max ${e["max-iterations"]} iterations, threshold ${(h*100).toFixed(0)}%`),e.auto&&this.log("Mode: auto (no interactive review)"),this.log("");let $=t.version,f=0,x=0;for(let r=1;r<=e["max-iterations"];r++){this.log(`
|
|
2
|
+
=== Iteration ${r} ===`),this.log("Running train evals...");let T=await this.runEvalsOnTarget(t,u,e),l=this.computePassRate(T);if(this.log(`Train pass rate: ${(l*100).toFixed(1)}%`),l>=h){this.log(`Train score ${(l*100).toFixed(1)}% >= threshold ${(h*100).toFixed(0)}%. Stopping.`);let o=await this.runEvalsOnTarget(t,v,e),E=this.computePassRate(o);this.log(`Final test score: ${(E*100).toFixed(1)}%`),await m("/api/optimization-runs",{evalTargetId:t.id,type:"content",iteration:r,inputVersion:t.version,trainScore:l,testScore:E,analysis:{reason:"threshold_reached"},status:"completed"});break}this.log("Analyzing results...");let S=T.map(o=>({evalName:o.evalName,prompt:o.prompt,expectations:o.expectations,grading:o.grading})),n=await A({content:t.content,evalResults:S,model:e["analyzer-model"]??e.model,timeout:e.timeout});this.log(`
|
|
3
|
+
Assessment: ${n.overallAssessment}`),this.log(`Suggestions: ${n.suggestions.length}`);for(let o of n.suggestions)this.log(` - [${o.section}] ${o.reasoning}`);if(n.proposedContent===t.content){this.log("No changes proposed. Stopping."),await m("/api/optimization-runs",{evalTargetId:t.id,type:"content",iteration:r,inputVersion:t.version,trainScore:l,analysis:{reason:"no_changes"},status:"completed"});break}if(!e.auto&&!await this.confirm(`
|
|
4
|
+
Apply the proposed changes as a new version? [y/N] `)){this.log("Changes not applied. Stopping."),await m("/api/optimization-runs",{evalTargetId:t.id,type:"content",iteration:r,inputVersion:t.version,trainScore:l,analysis:{reason:"user_declined",suggestions:n.suggestions},status:"completed"});break}let c=await R(`/api/eval-targets/${t.id}`,{content:n.proposedContent});this.log(`Created v${c.version}`),this.log("Running test evals...");let F=await this.runEvalsOnTarget(c,v,e),d=this.computePassRate(F);if(this.log(`Test pass rate: ${(d*100).toFixed(1)}%`),await m("/api/optimization-runs",{evalTargetId:t.id,type:"content",iteration:r,inputVersion:t.version,outputVersion:c.version,trainScore:l,testScore:d,analysis:{suggestions:n.suggestions,overallAssessment:n.overallAssessment},status:"completed"}),d>f)f=d,$=c.version,x=0;else if(x++,x>=2){this.log("No improvement for 2 consecutive iterations. Stopping.");break}t=c}this.log(`
|
|
5
|
+
=== Optimization Complete ===`),this.log(`Best version: v${$} (test score: ${(f*100).toFixed(1)}%)`)}async confirm(a){process.stdin.isTTY||this.error("Interactive approval requires a TTY. Re-run with --auto to apply changes without review.");let e=C({input:process.stdin,output:process.stdout});try{let t=await e.question(a);return/^y(es)?$/i.test(t.trim())}finally{e.close()}}async runEvalsOnTarget(a,e,t){let i=[];for(let s of e){let g=await z({prompt:s.prompt,targetContent:a.content,timeout:t.timeout,model:t.model});if(g.error){i.push({evalName:s.name,prompt:s.prompt,expectations:s.expectations,grading:{}});continue}if(s.expectations.length>0){let u=await k({prompt:s.prompt,output:g.output,expectations:s.expectations,model:t.model,timeout:t.timeout});i.push({evalName:s.name,prompt:s.prompt,expectations:s.expectations,grading:u})}else i.push({evalName:s.name,prompt:s.prompt,expectations:s.expectations,grading:{}})}return i}computePassRate(a){let e=a.filter(i=>Object.keys(i.grading).length>0);if(e.length===0)return 0;let t=e.map(i=>i.grading.summary?.pass_rate??0);return t.reduce((i,s)=>i+s,0)/t.length}};export{w as default};
|