@rulemetric/cli 0.12.8 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/base-command.js +1 -1
- package/dist/{chunk-CZTTW3MZ.js → chunk-2WQUOC6S.js} +20 -20
- package/dist/chunk-4VN5PG2S.js +1 -0
- package/dist/chunk-5MXZJWQS.js +2 -0
- package/dist/{chunk-NKJTNGXR.js → chunk-5XEGRU2R.js} +1 -1
- package/dist/chunk-6XTLP2G7.js +1 -0
- package/dist/chunk-74QIUVWY.js +1 -0
- package/dist/chunk-7ML2UCI7.js +1 -0
- package/dist/chunk-AHLVY52I.js +1 -0
- package/dist/{chunk-BTQQPQRG.js → chunk-AT2MMCOJ.js} +1 -1
- package/dist/{chunk-GOA5Q6TY.js → chunk-AXJAGWZB.js} +1 -1
- package/dist/{chunk-22ZDUL6H.js → chunk-BVOXWAJO.js} +1 -1
- package/dist/chunk-CT7FWSIU.js +3 -0
- package/dist/{chunk-RFISMWRW.js → chunk-DDYGA2DS.js} +1 -1
- package/dist/{chunk-OOFGV7EX.js → chunk-E5YE25QG.js} +1 -1
- package/dist/chunk-E7GQJML3.js +1 -0
- package/dist/{chunk-KJRMGTAO.js → chunk-E7RH43OS.js} +6 -6
- package/dist/{chunk-HQCPGDQE.js → chunk-ET7753IY.js} +1 -1
- package/dist/{chunk-INSKP5ED.js → chunk-EWGE7YTQ.js} +1 -1
- package/dist/chunk-EXVXKJ7Y.js +1 -0
- package/dist/chunk-FCTFHY3X.js +13 -0
- package/dist/chunk-FJX7WDPG.js +1 -0
- package/dist/{chunk-AGK5AUWT.js → chunk-FK55FXKQ.js} +1 -1
- package/dist/chunk-GH4EZK2O.js +1 -0
- package/dist/chunk-GN46S7T4.js +9 -0
- package/dist/{chunk-AFYZYXQ3.js → chunk-GQGGFHX7.js} +1 -1
- package/dist/chunk-GYNZ3OZF.js +1 -0
- package/dist/{chunk-GHWBW2IW.js → chunk-H4LUPIHJ.js} +1 -1
- package/dist/{chunk-YSLKV4XC.js → chunk-HAYO6JYI.js} +1 -1
- package/dist/{chunk-UKOI2Q4J.js → chunk-HFSISGCJ.js} +1 -1
- package/dist/chunk-HRJDKAJ4.js +1 -0
- package/dist/chunk-IC6VZHE6.js +1 -0
- package/dist/chunk-INAZLRY5.js +1 -0
- package/dist/{chunk-XFAFMYIP.js → chunk-IOX4VJPR.js} +2 -2
- package/dist/chunk-J4W6GNCF.js +2 -0
- package/dist/{chunk-PVPYLKB3.js → chunk-J7UZXRTH.js} +1 -1
- package/dist/{chunk-MILLBGCB.js → chunk-JKWB6UXA.js} +1 -1
- package/dist/chunk-JLBQKV4V.js +3 -0
- package/dist/{chunk-UTAETY5Z.js → chunk-JRAXWASH.js} +1 -1
- package/dist/{chunk-5PM4E27L.js → chunk-JRWQJZYQ.js} +1 -1
- package/dist/chunk-KINIFUZQ.js +1 -0
- package/dist/{chunk-E4RGYYBC.js → chunk-LDUJWJ6A.js} +1 -1
- package/dist/chunk-LMCICVE7.js +12 -0
- package/dist/chunk-MHVHGKDK.js +3 -0
- package/dist/chunk-MMM4B3KH.js +1 -0
- package/dist/{chunk-A2LHEUNW.js → chunk-MX34N7Y5.js} +1 -1
- package/dist/{chunk-TBU5SWA4.js → chunk-N26UEHI3.js} +1 -1
- package/dist/{chunk-DSJNZWCZ.js → chunk-N4RJ6CN7.js} +1 -1
- package/dist/{chunk-CPDMJGTC.js → chunk-N7SZWKCU.js} +1 -1
- package/dist/chunk-NRYVMGJN.js +1 -0
- package/dist/{chunk-FE2UWAIP.js → chunk-NSPDDSRR.js} +1 -1
- package/dist/{chunk-WCIX5KWG.js → chunk-O62OLKQL.js} +1 -1
- package/dist/{chunk-ARC4Q3GL.js → chunk-OLJF4K7L.js} +1 -1
- package/dist/{chunk-GRPUY2EZ.js → chunk-OMRH4JXS.js} +1 -1
- package/dist/{chunk-BJSVATVJ.js → chunk-OO2ZRKZY.js} +5 -5
- package/dist/chunk-OUK2AWEI.js +1 -0
- package/dist/chunk-OYAOLJUH.js +1 -0
- package/dist/chunk-PPAHIVG5.js +1 -0
- package/dist/{chunk-MJ5H7RDK.js → chunk-PPIZMPNI.js} +1 -1
- package/dist/chunk-PUX6P5KC.js +2 -0
- package/dist/chunk-QI3FJKGK.js +1 -0
- package/dist/chunk-RRM4HQG4.js +1 -0
- package/dist/{chunk-EZHL527S.js → chunk-S5CQSP35.js} +1 -1
- package/dist/chunk-S5U76YUI.js +1 -0
- package/dist/{chunk-NF2BIEAF.js → chunk-SBGXIPRN.js} +1 -1
- package/dist/chunk-SE3UKA3M.js +25 -0
- package/dist/chunk-SG4UDDHB.js +4 -0
- package/dist/chunk-SPJ73X7A.js +4 -0
- package/dist/chunk-SV73ASP2.js +1 -0
- package/dist/chunk-SWSOEYC7.js +1 -0
- package/dist/{chunk-DCH3QWB7.js → chunk-TTSDPDRH.js} +1 -1
- package/dist/chunk-TWPG3TEQ.js +48 -0
- package/dist/{chunk-AKXGUQ5B.js → chunk-U7BPWFOP.js} +1 -1
- package/dist/{chunk-L54TWWBY.js → chunk-UFQ43KXE.js} +1 -1
- package/dist/chunk-UJ7EPLKC.js +1 -0
- package/dist/{chunk-5N5S3NKS.js → chunk-UQDOOQCM.js} +1 -1
- package/dist/{chunk-G3HODHFN.js → chunk-V4PUYPTU.js} +1 -1
- package/dist/{chunk-SHYP27B7.js → chunk-VALRPZXH.js} +1 -1
- package/dist/{chunk-DPCAPIQB.js → chunk-VLWNSTYW.js} +1 -1
- package/dist/chunk-WSWZPNLG.js +1 -0
- package/dist/{chunk-UA5GEVGW.js → chunk-XAOLOR4Q.js} +1 -1
- package/dist/chunk-XAXEZI6Z.js +1 -0
- package/dist/{chunk-Z4YVOVMM.js → chunk-XFLKW6SJ.js} +1 -1
- package/dist/chunk-YNLHJHUP.js +1 -0
- package/dist/chunk-YP2CBUXH.js +5 -0
- package/dist/{chunk-VYZKKSX4.js → chunk-YUIRDBQQ.js} +1 -1
- package/dist/{chunk-2KJAM2NG.js → chunk-ZA6NXNPR.js} +1 -1
- package/dist/{chunk-UGACB66X.js → chunk-ZZHIBOMT.js} +1 -1
- package/dist/commands/antigravity/backfill.js +1 -1
- package/dist/commands/auth/create-key.js +1 -1
- package/dist/commands/auth/login.js +1 -1
- package/dist/commands/auth/logout.js +1 -1
- package/dist/commands/auth/revoke-key.js +1 -1
- package/dist/commands/auth/signup.js +1 -1
- package/dist/commands/auth/status.js +1 -1
- package/dist/commands/capture/scope.js +1 -1
- package/dist/commands/catalog/index.js +1 -1
- package/dist/commands/catalog/recommend.js +1 -1
- package/dist/commands/crons/list.js +1 -1
- package/dist/commands/crons/run.js +2 -2
- package/dist/commands/dashboard.js +1 -1
- package/dist/commands/diagnostics.js +1 -1
- package/dist/commands/eval-targets/create.js +1 -1
- package/dist/commands/eval-targets/list.js +1 -1
- package/dist/commands/evals/agent.js +3 -3
- package/dist/commands/evals/auto-run.js +1 -1
- package/dist/commands/evals/benchmark.js +1 -1
- package/dist/commands/evals/compare.js +1 -1
- package/dist/commands/evals/create.js +1 -1
- package/dist/commands/evals/optimize-description.js +1 -1
- package/dist/commands/evals/optimize.js +1 -1
- package/dist/commands/evals/refine-cases.js +1 -1
- package/dist/commands/evals/run.js +1 -1
- package/dist/commands/evals/validate-judge.js +1 -1
- package/dist/commands/experiments/list.js +1 -1
- package/dist/commands/gateway/ensure.js +1 -1
- package/dist/commands/hooks/emit-instructions.js +1 -1
- package/dist/commands/hooks/install.js +1 -1
- package/dist/commands/hooks/run.js +1 -1
- package/dist/commands/hooks/uninstall.js +1 -1
- package/dist/commands/instructions/create.js +1 -1
- package/dist/commands/instructions/dedupe.js +1 -1
- package/dist/commands/instructions/delete.js +1 -1
- package/dist/commands/instructions/fork.js +1 -1
- package/dist/commands/instructions/get.js +1 -1
- package/dist/commands/instructions/list.js +1 -1
- package/dist/commands/instructions/promote.js +1 -1
- package/dist/commands/instructions/pull.js +1 -1
- package/dist/commands/instructions/restore.js +1 -1
- package/dist/commands/instructions/retract.js +1 -1
- package/dist/commands/instructions/upstream.js +1 -1
- package/dist/commands/instructions/versions.js +1 -1
- package/dist/commands/launcher.js +1 -1
- package/dist/commands/local/down.js +1 -1
- package/dist/commands/local/up.js +1 -1
- package/dist/commands/org/current.js +1 -1
- package/dist/commands/org/list.js +1 -1
- package/dist/commands/org/switch.js +1 -1
- package/dist/commands/packs/add-instruction.js +1 -1
- package/dist/commands/packs/apply.js +1 -1
- package/dist/commands/packs/create.js +1 -1
- package/dist/commands/packs/fork.js +1 -1
- package/dist/commands/packs/get.js +1 -1
- package/dist/commands/packs/list.js +1 -1
- package/dist/commands/packs/remove-instruction.js +1 -1
- package/dist/commands/proxy/env.js +1 -1
- package/dist/commands/proxy/logs.js +1 -1
- package/dist/commands/proxy/restart.js +1 -1
- package/dist/commands/proxy/setup.js +1 -1
- package/dist/commands/proxy/start.js +1 -1
- package/dist/commands/proxy/status.js +1 -1
- package/dist/commands/proxy/stop.js +1 -1
- package/dist/commands/research/backfill.js +1 -1
- package/dist/commands/research/tail-transcript.js +1 -1
- package/dist/commands/service/install.js +18 -18
- package/dist/commands/service/status.js +1 -1
- package/dist/commands/service/uninstall.js +1 -1
- package/dist/commands/sessions/analyze.js +1 -1
- package/dist/commands/sessions/delete.js +1 -1
- package/dist/commands/sessions/end.js +1 -1
- package/dist/commands/sessions/import.js +1 -1
- package/dist/commands/sessions/list.js +1 -1
- package/dist/commands/sessions/start.js +1 -1
- package/dist/commands/setup.js +1 -1
- package/dist/commands/skills/export-approved.js +1 -1
- package/dist/commands/skills/install.js +1 -1
- package/dist/commands/skills/list.js +1 -1
- package/dist/commands/skills/publish.js +1 -1
- package/dist/commands/skills/search.js +1 -1
- package/dist/commands/skills/update.js +1 -1
- package/dist/commands/suggestions/accept.js +1 -1
- package/dist/commands/suggestions/auto.js +1 -1
- package/dist/commands/suggestions/dismiss.js +1 -1
- package/dist/commands/suggestions/list.js +1 -1
- package/dist/commands/suggestions/refresh.js +1 -1
- package/dist/commands/suggestions/undo.js +1 -1
- package/dist/dashboard/Dashboard.js +1 -1
- package/dist/dashboard/data.js +1 -1
- package/dist/dashboard/launcher/LauncherOverlay.js +1 -1
- package/dist/dashboard/launcher/api.js +1 -1
- package/dist/dashboard/launcher/useLaunchJobs.js +1 -1
- package/dist/dist-NHS7HY5Q.js +1 -0
- package/dist/harbor/cost_delta.py +377 -0
- package/dist/harbor/mcp_container_shell.py +156 -0
- package/dist/harbor/provenance.py +361 -0
- package/dist/harbor/rm-batch-task/environment/Dockerfile +6 -0
- package/dist/harbor/rm-batch-task/environment/items.jsonl +60 -0
- package/dist/harbor/rm-batch-task/inject.md +8 -0
- package/dist/harbor/rm-batch-task/instruction.md +8 -0
- package/dist/harbor/rm-batch-task/solution/solve.sh +68 -0
- package/dist/harbor/rm-batch-task/task.toml +30 -0
- package/dist/harbor/rm-batch-task/tests/test.sh +23 -0
- package/dist/harbor/rm-batch-task/tests/test_outputs.py +32 -0
- package/dist/harbor/rm-locality-task/environment/Dockerfile +8 -0
- package/dist/harbor/rm-locality-task/environment/workspace/deploy/prod/ingest.yaml +3 -0
- package/dist/harbor/rm-locality-task/environment/workspace/deploy/staging/ingest.yaml +3 -0
- package/dist/harbor/rm-locality-task/environment/workspace/docs/architecture/ingest.md +6 -0
- package/dist/harbor/rm-locality-task/environment/workspace/legacy/ingest_v1/config.yaml +2 -0
- package/dist/harbor/rm-locality-task/environment/workspace/packages/shared/defaults.py +4 -0
- package/dist/harbor/rm-locality-task/environment/workspace/services/ingest/config/limits.defaults.yaml +4 -0
- package/dist/harbor/rm-locality-task/environment/workspace/services/ingest/config/limits.yaml +3 -0
- package/dist/harbor/rm-locality-task/environment/workspace/services/ingest/loader.py +15 -0
- package/dist/harbor/rm-locality-task/environment/workspace/tests/fixtures/ingest_config.yaml +3 -0
- package/dist/harbor/rm-locality-task/inject.md +12 -0
- package/dist/harbor/rm-locality-task/instruction.md +4 -0
- package/dist/harbor/rm-locality-task/solution/solve.sh +14 -0
- package/dist/harbor/rm-locality-task/task.toml +30 -0
- package/dist/harbor/rm-locality-task/tests/test.sh +23 -0
- package/dist/harbor/rm-locality-task/tests/test_outputs.py +19 -0
- package/dist/harbor/rm-migrations-task/environment/Dockerfile +7 -0
- package/dist/harbor/rm-migrations-task/environment/applied.txt +2 -0
- package/dist/harbor/rm-migrations-task/environment/migrations/00001_init.sql +1 -0
- package/dist/harbor/rm-migrations-task/environment/migrations/00002_add_email.sql +1 -0
- package/dist/harbor/rm-migrations-task/inject.md +7 -0
- package/dist/harbor/rm-migrations-task/instruction.md +8 -0
- package/dist/harbor/rm-migrations-task/solution/solve.sh +8 -0
- package/dist/harbor/rm-migrations-task/task.toml +30 -0
- package/dist/harbor/rm-migrations-task/tests/test.sh +23 -0
- package/dist/harbor/rm-migrations-task/tests/test_outputs.py +36 -0
- package/dist/harbor/rm-pagination-task/environment/Dockerfile +6 -0
- package/dist/harbor/rm-pagination-task/environment/store.py +29 -0
- package/dist/harbor/rm-pagination-task/inject.md +8 -0
- package/dist/harbor/rm-pagination-task/instruction.md +5 -0
- package/dist/harbor/rm-pagination-task/solution/solve.sh +7 -0
- package/dist/harbor/rm-pagination-task/task.toml +30 -0
- package/dist/harbor/rm-pagination-task/tests/test.sh +23 -0
- package/dist/harbor/rm-pagination-task/tests/test_outputs.py +14 -0
- package/dist/harbor/rm_host_claude.py +264 -0
- package/dist/harbor/rm_host_codex.py +208 -0
- package/dist/harbor/run.sh +268 -0
- package/dist/harbor/summarize.py +248 -0
- package/dist/lib/active-org-refresh.js +1 -1
- package/dist/lib/agent-loop.js +1 -1
- package/dist/lib/antigravity-transcript.js +1 -1
- package/dist/lib/api-client.js +1 -1
- package/dist/lib/capture-scope-config.js +1 -1
- package/dist/lib/codex-capture-scope.js +1 -1
- package/dist/lib/codex-exec.js +1 -1
- package/dist/lib/ensure-api-key.js +1 -1
- package/dist/lib/ensure-fresh-token.js +1 -1
- package/dist/lib/eval-accumulation.js +1 -0
- package/dist/lib/eval-analyzer.js +1 -1
- package/dist/lib/eval-benchmark.js +1 -1
- package/dist/lib/eval-candidate-run.js +1 -1
- package/dist/lib/eval-case-generator.js +1 -1
- package/dist/lib/eval-conversation-handler.js +1 -1
- package/dist/lib/eval-executor.js +1 -1
- package/dist/lib/eval-grader.js +1 -1
- package/dist/lib/eval-meta-judge.js +1 -1
- package/dist/lib/eval-resume.js +1 -1
- package/dist/lib/eval-run-batch.js +1 -1
- package/dist/lib/eval-trigger-tester.js +1 -1
- package/dist/lib/evolution-enrolment.js +1 -1
- package/dist/lib/harbor-readiness.js +1 -0
- package/dist/lib/hooks-capture-gate.js +1 -1
- package/dist/lib/insights/generation/installed-asks.js +1 -1
- package/dist/lib/insights/nomination/nominate.js +1 -1
- package/dist/lib/instruction-proposer.js +1 -1
- package/dist/lib/instruction-snapshot.js +1 -1
- package/dist/lib/instructions/acceptance/accept-events.js +1 -1
- package/dist/lib/instructions/acceptance/accept-hook.js +1 -1
- package/dist/lib/instructions/acceptance/agent-instruction-files.js +1 -0
- package/dist/lib/instructions/acceptance/hook-install.js +1 -1
- package/dist/lib/instructions/acceptance/reverse-applications.js +1 -1
- package/dist/lib/instructions/acceptance/suggestion-accept.js +1 -1
- package/dist/lib/jobs/handlers/cron-auto-accept-suggestions.js +1 -1
- package/dist/lib/jobs/handlers/cron-eval-autorun.js +1 -1
- package/dist/lib/jobs/handlers/cron-instruction-evolution.js +1 -1
- package/dist/lib/jobs/handlers/cron-instruction-training.js +1 -1
- package/dist/lib/jobs/handlers/cron-reconcile-attribution.js +1 -1
- package/dist/lib/jobs/handlers/cron-refine-cases.js +1 -1
- package/dist/lib/jobs/handlers/cron-refresh-pricing.js +1 -1
- package/dist/lib/jobs/handlers/cron-retire-unused-skills.js +1 -1
- package/dist/lib/jobs/handlers/cron-suggest-instructions.js +1 -1
- package/dist/lib/jobs/handlers/distill-memories.js +1 -1
- package/dist/lib/jobs/handlers/harbor-ab-verdict.js +1 -0
- package/dist/lib/jobs/handlers/harbor-ab.js +1 -0
- package/dist/lib/jobs/handlers/process-changelog.js +1 -1
- package/dist/lib/jobs/handlers/process-conversation.js +1 -1
- package/dist/lib/jobs/handlers/process-eval.js +1 -1
- package/dist/lib/jobs/handlers/process-insights.js +1 -1
- package/dist/lib/jobs/handlers/process-judge-review.js +1 -1
- package/dist/lib/jobs/handlers/process-launch.js +1 -1
- package/dist/lib/jobs/handlers/process-memory-ab.js +1 -1
- package/dist/lib/jobs/handlers/process-run-cleanup.js +1 -1
- package/dist/lib/jobs/handlers/process-session-goal.js +1 -1
- package/dist/lib/jobs/shed-retry.js +1 -0
- package/dist/lib/judge-review-runner.js +1 -1
- package/dist/lib/llm-client.js +1 -1
- package/dist/lib/llm-json.js +1 -0
- package/dist/lib/llm-reply-parsers.js +1 -0
- package/dist/lib/machine-pause-detector.js +1 -1
- package/dist/lib/manual-tasks-meta.js +1 -1
- package/dist/lib/manual-tasks.js +1 -1
- package/dist/lib/measurement-llm.js +1 -0
- package/dist/lib/pi-exec.js +1 -0
- package/dist/lib/projects/workspace/eval-workspace.js +1 -1
- package/dist/lib/projects/workspace/worktree.js +1 -1
- package/dist/lib/refine-cases.js +1 -1
- package/dist/lib/research-client.js +1 -1
- package/dist/lib/retraction.js +1 -1
- package/dist/lib/reviewer-detect.js +1 -1
- package/dist/lib/service-environment.js +1 -0
- package/dist/lib/setup-steps.js +1 -1
- package/dist/lib/telemetry.js +1 -1
- package/dist/lib/terminal-target-heartbeat.js +1 -1
- package/dist/lib/tool-bin.js +1 -0
- package/dist/lib/tool-cli-map.js +1 -0
- package/dist/lib/transcript-instruction-linking.js +1 -0
- package/dist/lib/transcript-sync.js +1 -1
- package/dist/lib/work-report.js +1 -0
- package/oclif.manifest.json +17 -4
- package/package.json +10 -8
- package/dist/chunk-3DZD322E.js +0 -1
- package/dist/chunk-5OX6YKTR.js +0 -5
- package/dist/chunk-APXQLQZY.js +0 -1
- package/dist/chunk-C6GLNE6X.js +0 -1
- package/dist/chunk-CNQ5DTNC.js +0 -1
- package/dist/chunk-E7XOYGL5.js +0 -20
- package/dist/chunk-EVD5A3VI.js +0 -1
- package/dist/chunk-FUYR7CFD.js +0 -1
- package/dist/chunk-GN3ZHQ46.js +0 -1
- package/dist/chunk-GWDZNGUP.js +0 -2
- package/dist/chunk-LUCVMLHV.js +0 -1
- package/dist/chunk-O65AQZYB.js +0 -1
- package/dist/chunk-PWTTCSQA.js +0 -1
- package/dist/chunk-QBNLZJIR.js +0 -6
- package/dist/chunk-QXPYGN7I.js +0 -1
- package/dist/chunk-RIVEBBZC.js +0 -3
- package/dist/chunk-RLEMDKQ4.js +0 -12
- package/dist/chunk-S2SKT5PV.js +0 -1
- package/dist/chunk-TZA4HAVF.js +0 -1
- package/dist/chunk-UUFCTEBT.js +0 -1
- package/dist/chunk-UXB6DRR7.js +0 -3
- package/dist/chunk-W23MESHM.js +0 -1
- package/dist/chunk-W67DO5PT.js +0 -1
- package/dist/chunk-WMIUFMDI.js +0 -48
- package/dist/chunk-WO6MS654.js +0 -13
- package/dist/chunk-WTMJSL7R.js +0 -1
- package/dist/dist-XEWF3R3N.js +0 -1
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Dependency-free MCP stdio server exposing one tool: `bash`, executed
|
|
2
|
+
INSIDE the Harbor trial container.
|
|
3
|
+
|
|
4
|
+
This is a tool proxy, not a network proxy: the host Claude Code's own Bash
|
|
5
|
+
tool is disabled by the agent, and every shell command is routed here, run
|
|
6
|
+
via `docker exec` in the task container. Around each command the host
|
|
7
|
+
workdir and the container's /app are synced with tar pipes (docker cp has no
|
|
8
|
+
excludes; the injected CLAUDE.md and .claude/ must never reach the
|
|
9
|
+
container).
|
|
10
|
+
|
|
11
|
+
Config via env: RM_CONTAINER_ID (docker container id), RM_WORKDIR (host dir
|
|
12
|
+
mirroring /app).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import subprocess
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
CONTAINER = os.environ["RM_CONTAINER_ID"]
|
|
21
|
+
WORKDIR = os.environ["RM_WORKDIR"]
|
|
22
|
+
SYNC_EXCLUDES = ["CLAUDE.md", ".claude"]
|
|
23
|
+
DEFAULT_TIMEOUT = 120
|
|
24
|
+
MAX_OUTPUT = 30_000
|
|
25
|
+
|
|
26
|
+
TOOL = {
|
|
27
|
+
"name": "bash",
|
|
28
|
+
"description": (
|
|
29
|
+
"Run a shell command inside the task container (bash -lc). This is "
|
|
30
|
+
"the ONLY shell for this task: the host Bash tool is disabled. Files "
|
|
31
|
+
"edited with file tools are synced into the container before the "
|
|
32
|
+
"command runs, and container file changes are synced back after."
|
|
33
|
+
),
|
|
34
|
+
"inputSchema": {
|
|
35
|
+
"type": "object",
|
|
36
|
+
"properties": {
|
|
37
|
+
"command": {"type": "string", "description": "Shell command to run"},
|
|
38
|
+
"timeout_sec": {
|
|
39
|
+
"type": "number",
|
|
40
|
+
"description": f"Seconds before the command is killed (default {DEFAULT_TIMEOUT})",
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
"required": ["command"],
|
|
44
|
+
},
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _sync_to_container() -> None:
|
|
49
|
+
tar = subprocess.Popen(
|
|
50
|
+
["tar", "-C", WORKDIR]
|
|
51
|
+
+ [f"--exclude=./{e}" for e in SYNC_EXCLUDES]
|
|
52
|
+
+ ["-cf", "-", "."],
|
|
53
|
+
stdout=subprocess.PIPE,
|
|
54
|
+
)
|
|
55
|
+
subprocess.run(
|
|
56
|
+
["docker", "exec", "-i", CONTAINER, "tar", "-C", "/app", "-xf", "-"],
|
|
57
|
+
stdin=tar.stdout,
|
|
58
|
+
check=True,
|
|
59
|
+
capture_output=True,
|
|
60
|
+
)
|
|
61
|
+
tar.wait()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _sync_from_container() -> None:
|
|
65
|
+
tar = subprocess.Popen(
|
|
66
|
+
["docker", "exec", CONTAINER, "tar", "-C", "/app", "-cf", "-", "."],
|
|
67
|
+
stdout=subprocess.PIPE,
|
|
68
|
+
)
|
|
69
|
+
subprocess.run(
|
|
70
|
+
["tar", "-C", WORKDIR, "-xf", "-"],
|
|
71
|
+
stdin=tar.stdout,
|
|
72
|
+
check=True,
|
|
73
|
+
capture_output=True,
|
|
74
|
+
)
|
|
75
|
+
tar.wait()
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _run_bash(command: str, timeout_sec: float) -> str:
|
|
79
|
+
_sync_to_container()
|
|
80
|
+
try:
|
|
81
|
+
proc = subprocess.run(
|
|
82
|
+
["docker", "exec", "-w", "/app", CONTAINER, "bash", "-lc", command],
|
|
83
|
+
capture_output=True,
|
|
84
|
+
text=True,
|
|
85
|
+
timeout=timeout_sec,
|
|
86
|
+
)
|
|
87
|
+
out = proc.stdout[-MAX_OUTPUT:]
|
|
88
|
+
err = proc.stderr[-MAX_OUTPUT:]
|
|
89
|
+
result = f"[exit_code] {proc.returncode}"
|
|
90
|
+
if out:
|
|
91
|
+
result = f"[stdout]\n{out}\n{result}"
|
|
92
|
+
if err:
|
|
93
|
+
result = f"{result}\n[stderr]\n{err}"
|
|
94
|
+
except subprocess.TimeoutExpired:
|
|
95
|
+
result = f"[error] command timed out after {timeout_sec}s"
|
|
96
|
+
_sync_from_container()
|
|
97
|
+
return result
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _respond(msg_id, result=None, error=None) -> None:
|
|
101
|
+
resp = {"jsonrpc": "2.0", "id": msg_id}
|
|
102
|
+
if error is not None:
|
|
103
|
+
resp["error"] = error
|
|
104
|
+
else:
|
|
105
|
+
resp["result"] = result
|
|
106
|
+
sys.stdout.write(json.dumps(resp) + "\n")
|
|
107
|
+
sys.stdout.flush()
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def main() -> None:
|
|
111
|
+
for line in sys.stdin:
|
|
112
|
+
line = line.strip()
|
|
113
|
+
if not line:
|
|
114
|
+
continue
|
|
115
|
+
msg = json.loads(line)
|
|
116
|
+
method, msg_id = msg.get("method"), msg.get("id")
|
|
117
|
+
if method == "initialize":
|
|
118
|
+
_respond(
|
|
119
|
+
msg_id,
|
|
120
|
+
{
|
|
121
|
+
"protocolVersion": msg["params"].get(
|
|
122
|
+
"protocolVersion", "2024-11-05"
|
|
123
|
+
),
|
|
124
|
+
"capabilities": {"tools": {}},
|
|
125
|
+
"serverInfo": {"name": "container", "version": "0.1.0"},
|
|
126
|
+
},
|
|
127
|
+
)
|
|
128
|
+
elif method == "tools/list":
|
|
129
|
+
_respond(msg_id, {"tools": [TOOL]})
|
|
130
|
+
elif method == "tools/call":
|
|
131
|
+
args = msg["params"].get("arguments", {})
|
|
132
|
+
try:
|
|
133
|
+
text = _run_bash(
|
|
134
|
+
args["command"],
|
|
135
|
+
float(args.get("timeout_sec") or DEFAULT_TIMEOUT),
|
|
136
|
+
)
|
|
137
|
+
_respond(
|
|
138
|
+
msg_id,
|
|
139
|
+
{"content": [{"type": "text", "text": text}], "isError": False},
|
|
140
|
+
)
|
|
141
|
+
except Exception as exc: # sync/docker failures reach the model as text
|
|
142
|
+
_respond(
|
|
143
|
+
msg_id,
|
|
144
|
+
{
|
|
145
|
+
"content": [{"type": "text", "text": f"[proxy error] {exc}"}],
|
|
146
|
+
"isError": True,
|
|
147
|
+
},
|
|
148
|
+
)
|
|
149
|
+
elif method == "ping":
|
|
150
|
+
_respond(msg_id, {})
|
|
151
|
+
elif msg_id is not None: # unknown request; notifications are ignored
|
|
152
|
+
_respond(msg_id, error={"code": -32601, "message": f"unknown: {method}"})
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
if __name__ == "__main__":
|
|
156
|
+
main()
|
|
@@ -0,0 +1,361 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""What harness measured a Harbor job — and whether two jobs are comparable.
|
|
3
|
+
|
|
4
|
+
Two arms are a valid A/B only if the only thing that differed between them is
|
|
5
|
+
the instruction. Everything else that CAN differ and change the numbers is a
|
|
6
|
+
pooling key here, recorded per trial and checked before any verdict.
|
|
7
|
+
|
|
8
|
+
Origin (2026-09-02). The 2026-08-31 scorecard recorded `rm-pagination-task`
|
|
9
|
+
baseline at 2/10; the 2026-09-02 re-measurement got 25/25. The cause was
|
|
10
|
+
`--allowedTools Bash` landing in #415 between them: the "failures" were largely
|
|
11
|
+
the agent stalling on an unapproved Bash command, not computing a wrong
|
|
12
|
+
aggregate. Nothing in either job dir recorded the regime, so for two days a
|
|
13
|
+
HARNESS change read as the rule having stopped working. scripts/harbor/README.md
|
|
14
|
+
item 7 already stated the rule -- "runs made under different permission regimes
|
|
15
|
+
are different harnesses" -- as prose, and prose does not refuse.
|
|
16
|
+
|
|
17
|
+
This module is the single owner of that rule. `cost_delta.py` and
|
|
18
|
+
`verdict_bridge.py` import it; the scheduled TypeScript path SHELLS OUT to
|
|
19
|
+
`--compare` rather than re-deriving POOLING_KEYS, because a hand-copied
|
|
20
|
+
predicate is how the canonical-checkout, judge-segment and auto-accept defects
|
|
21
|
+
each shipped (CLAUDE.md hard invariants).
|
|
22
|
+
|
|
23
|
+
Stdlib only, and importable without `harbor` installed, so the contract test can
|
|
24
|
+
exercise every path on synthetic job dirs.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import json
|
|
31
|
+
import os
|
|
32
|
+
import sys
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Any, Iterable, Sequence
|
|
35
|
+
|
|
36
|
+
# The keys that make two runs different harnesses. A difference on any of these
|
|
37
|
+
# means the arms are not an A/B of the instruction, whatever the numbers say.
|
|
38
|
+
POOLING_KEYS = ("engine", "engine_version", "model", "permission_regime", "billing")
|
|
39
|
+
|
|
40
|
+
# Flags that define the permission regime, in a FIXED order so the regime string
|
|
41
|
+
# does not change when the command builder reorders its argv. Read from the
|
|
42
|
+
# actual argv rather than a constant in this file: a regime hardcoded here could
|
|
43
|
+
# never notice the change it exists to catch.
|
|
44
|
+
_REGIME_VALUE_FLAGS = (
|
|
45
|
+
"--permission-mode",
|
|
46
|
+
"--allowedTools",
|
|
47
|
+
"--sandbox",
|
|
48
|
+
"--ask-for-approval",
|
|
49
|
+
"--approval-mode",
|
|
50
|
+
)
|
|
51
|
+
_REGIME_BOOL_FLAGS = (
|
|
52
|
+
"--dangerously-skip-permissions",
|
|
53
|
+
"--full-auto",
|
|
54
|
+
"--yolo",
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
# Presence of a credential is a FACT about the run and belongs in the stamp; the
|
|
58
|
+
# credential itself never does. Harbor records agent env verbatim in each trial's
|
|
59
|
+
# config.json, so anything written here is written next to that.
|
|
60
|
+
_BYOK_VARS = ("ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN")
|
|
61
|
+
|
|
62
|
+
STAMP_FILENAME = "rm-provenance.json"
|
|
63
|
+
FORMAT_VERSION = 1
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _flag_value(argv: Sequence[str], flag: str) -> str | None:
|
|
67
|
+
"""The value following `flag`, or None. Supports `--flag=value` too."""
|
|
68
|
+
for i, token in enumerate(argv):
|
|
69
|
+
if token == flag:
|
|
70
|
+
return argv[i + 1] if i + 1 < len(argv) else None
|
|
71
|
+
if token.startswith(flag + "="):
|
|
72
|
+
return token.split("=", 1)[1]
|
|
73
|
+
return None
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def permission_regime(argv: Sequence[str]) -> str:
|
|
77
|
+
"""The regime as a stable string, e.g. 'acceptEdits+Bash'.
|
|
78
|
+
|
|
79
|
+
'unspecified' when the argv names no regime at all -- which is itself a
|
|
80
|
+
distinguishing fact, not a missing value: a run with no stated regime and a
|
|
81
|
+
run with `acceptEdits` are different harnesses.
|
|
82
|
+
"""
|
|
83
|
+
parts: list[str] = []
|
|
84
|
+
for flag in _REGIME_VALUE_FLAGS:
|
|
85
|
+
value = _flag_value(argv, flag)
|
|
86
|
+
if value:
|
|
87
|
+
# A comma list is normalised so tool ORDER cannot fork the string.
|
|
88
|
+
parts.append("+".join(sorted(v.strip() for v in value.split(",") if v.strip())))
|
|
89
|
+
parts.extend(flag.lstrip("-") for flag in _REGIME_BOOL_FLAGS if flag in argv)
|
|
90
|
+
return "+".join(parts) if parts else "unspecified"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def stamp(
|
|
94
|
+
argv: Sequence[str],
|
|
95
|
+
*,
|
|
96
|
+
observed_model: str | None = None,
|
|
97
|
+
engine: str | None = None,
|
|
98
|
+
engine_version: str | None = None,
|
|
99
|
+
env: dict[str, str] | None = None,
|
|
100
|
+
delivery: Iterable[str] | None = None,
|
|
101
|
+
) -> dict[str, Any]:
|
|
102
|
+
"""The per-trial provenance record, written beside the agent's output.
|
|
103
|
+
|
|
104
|
+
`observed_model` is what the run REPORTS having billed (claude's
|
|
105
|
+
`modelUsage`); it outranks the `--model` flag, because a credit-exhaustion
|
|
106
|
+
fallback silently swaps the model while the flag still says haiku -- the
|
|
107
|
+
same failure that orphaned the judge segment (CLAUDE.md). When nothing
|
|
108
|
+
observed the model, that is stated (`model_source: 'requested'`) rather than
|
|
109
|
+
presented as if it had been.
|
|
110
|
+
"""
|
|
111
|
+
argv = list(argv)
|
|
112
|
+
environ = os.environ if env is None else env
|
|
113
|
+
requested = _flag_value(argv, "--model") or _flag_value(argv, "-m")
|
|
114
|
+
if engine is None and argv:
|
|
115
|
+
engine = Path(argv[0]).name or None
|
|
116
|
+
return {
|
|
117
|
+
"format_version": FORMAT_VERSION,
|
|
118
|
+
"engine": engine,
|
|
119
|
+
"engine_version": engine_version,
|
|
120
|
+
"model": observed_model or requested,
|
|
121
|
+
"model_requested": requested,
|
|
122
|
+
"model_source": "observed" if observed_model else ("requested" if requested else "unknown"),
|
|
123
|
+
"permission_regime": permission_regime(argv),
|
|
124
|
+
"billing": "byok" if any(environ.get(v) for v in _BYOK_VARS) else "subscription",
|
|
125
|
+
"delivery": sorted(delivery) if delivery else [],
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def observed_model_from_claude(output: str | bytes) -> str | None:
|
|
130
|
+
"""The model(s) a `claude -p --output-format json` envelope reports billing.
|
|
131
|
+
|
|
132
|
+
None when the envelope says nothing -- unknown, never a guess. Two keys is
|
|
133
|
+
NOT one model: the run fell back mid-flight, and a stamp that reported only
|
|
134
|
+
the first would call that the same harness as a clean single-model run, which
|
|
135
|
+
is the credit-exhaustion shape that orphaned the judge segment.
|
|
136
|
+
"""
|
|
137
|
+
if isinstance(output, bytes):
|
|
138
|
+
output = output.decode("utf-8", errors="replace")
|
|
139
|
+
try:
|
|
140
|
+
envelope = json.loads(output)
|
|
141
|
+
except (ValueError, TypeError):
|
|
142
|
+
return None
|
|
143
|
+
usage = envelope.get("modelUsage") if isinstance(envelope, dict) else None
|
|
144
|
+
if not isinstance(usage, dict) or not usage:
|
|
145
|
+
return None
|
|
146
|
+
return "+".join(sorted(str(k) for k in usage))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
_CLI_VERSION_CACHE: dict[str, str | None] = {}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def cli_version(binary: str) -> str | None:
|
|
153
|
+
"""`<binary> --version`, cached per process. None when it cannot be read.
|
|
154
|
+
|
|
155
|
+
This is the version that CHANGES under a scheduled run on someone else's
|
|
156
|
+
machine, which is what makes it a pooling key -- unlike the agent module's
|
|
157
|
+
own version, recorded separately as `agent_version`.
|
|
158
|
+
"""
|
|
159
|
+
if binary not in _CLI_VERSION_CACHE:
|
|
160
|
+
import subprocess # local: keeps the module importable in odd sandboxes
|
|
161
|
+
|
|
162
|
+
try:
|
|
163
|
+
proc = subprocess.run(
|
|
164
|
+
[binary, "--version"], capture_output=True, text=True, timeout=20
|
|
165
|
+
)
|
|
166
|
+
out = (proc.stdout or proc.stderr or "").strip()
|
|
167
|
+
_CLI_VERSION_CACHE[binary] = out.splitlines()[0].strip() if out else None
|
|
168
|
+
except (OSError, subprocess.SubprocessError):
|
|
169
|
+
_CLI_VERSION_CACHE[binary] = None
|
|
170
|
+
return _CLI_VERSION_CACHE[binary]
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def write_stamp(agent_dir: Path, record: dict[str, Any]) -> Path:
|
|
174
|
+
"""Write a stamp beside a trial's agent output. Never raises into the run.
|
|
175
|
+
|
|
176
|
+
A run that produced real work must not be destroyed by a failure to record
|
|
177
|
+
what produced it; an unstamped job is refused later by name, which is the
|
|
178
|
+
safe direction.
|
|
179
|
+
"""
|
|
180
|
+
path = Path(agent_dir) / STAMP_FILENAME
|
|
181
|
+
try:
|
|
182
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
183
|
+
path.write_text(json.dumps(record, indent=2, sort_keys=True))
|
|
184
|
+
except OSError as exc: # pragma: no cover - defensive
|
|
185
|
+
print(f"provenance: could not write {path}: {exc}", file=sys.stderr)
|
|
186
|
+
return path
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def read_job(job_dir: str | Path) -> dict[str, Any]:
|
|
190
|
+
"""Merge a job's per-trial stamps into one, or report which key disagrees.
|
|
191
|
+
|
|
192
|
+
A job whose own trials disagree is NOT a job: its mean already pools two
|
|
193
|
+
harnesses, so no single stamp can describe it and `provenance` is None.
|
|
194
|
+
"""
|
|
195
|
+
job_dir = Path(job_dir)
|
|
196
|
+
values: dict[str, set[str]] = {k: set() for k in POOLING_KEYS}
|
|
197
|
+
stamped = 0
|
|
198
|
+
for path in sorted(job_dir.rglob(STAMP_FILENAME)):
|
|
199
|
+
try:
|
|
200
|
+
record = json.loads(path.read_text())
|
|
201
|
+
except (OSError, ValueError):
|
|
202
|
+
continue
|
|
203
|
+
if not isinstance(record, dict):
|
|
204
|
+
continue
|
|
205
|
+
stamped += 1
|
|
206
|
+
for key in POOLING_KEYS:
|
|
207
|
+
value = record.get(key)
|
|
208
|
+
if value is not None:
|
|
209
|
+
values[key].add(str(value))
|
|
210
|
+
|
|
211
|
+
mixed = [k for k in POOLING_KEYS if len(values[k]) > 1]
|
|
212
|
+
provenance: dict[str, Any] | None = None
|
|
213
|
+
if stamped and not mixed:
|
|
214
|
+
provenance = {k: (next(iter(values[k])) if values[k] else None) for k in POOLING_KEYS}
|
|
215
|
+
return {"provenance": provenance, "mixed": mixed, "trials_stamped": stamped}
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def compare(
|
|
219
|
+
baseline_dir: str | Path,
|
|
220
|
+
with_dir: str | Path,
|
|
221
|
+
*,
|
|
222
|
+
allow_cross: Iterable[str] = (),
|
|
223
|
+
allow_unstamped: bool = False,
|
|
224
|
+
) -> dict[str, Any]:
|
|
225
|
+
"""Are these two job dirs an A/B of the instruction, or of the harness?
|
|
226
|
+
|
|
227
|
+
`allow_cross` is the typed exception: a key listed there stops refusing and
|
|
228
|
+
moves to `pooled_across`, which every caller carries into its verdict, so a
|
|
229
|
+
cross-harness number can never be quoted as a within-harness one.
|
|
230
|
+
|
|
231
|
+
`allow_unstamped` exists only for job dirs measured before stamps existed.
|
|
232
|
+
It says "I could not verify", never "I verified they match" -- the reason
|
|
233
|
+
stays on the result either way, so an unverified verdict cannot be quoted as
|
|
234
|
+
a verified one.
|
|
235
|
+
"""
|
|
236
|
+
allow = [k for k in allow_cross if k]
|
|
237
|
+
unknown = [k for k in allow if k not in POOLING_KEYS]
|
|
238
|
+
arms = {"baseline": read_job(baseline_dir), "with-arm": read_job(with_dir)}
|
|
239
|
+
|
|
240
|
+
reasons: list[str] = []
|
|
241
|
+
unstamped = [label for label, a in arms.items() if a["provenance"] is None and not a["mixed"]]
|
|
242
|
+
mixed_arms = [label for label, a in arms.items() if a["mixed"]]
|
|
243
|
+
|
|
244
|
+
for key in unknown:
|
|
245
|
+
reasons.append(f"--allow-cross named '{key}', which is not a pooling key ({', '.join(POOLING_KEYS)})")
|
|
246
|
+
for label in unstamped:
|
|
247
|
+
reasons.append(
|
|
248
|
+
f"{label} carries no provenance stamp — unknown provenance is not matching provenance, "
|
|
249
|
+
+ ("so this verdict is UNVERIFIED (--allow-unstamped)" if allow_unstamped
|
|
250
|
+
else "so this pair is refused rather than assumed comparable")
|
|
251
|
+
)
|
|
252
|
+
for label in mixed_arms:
|
|
253
|
+
keys = ", ".join(arms[label]["mixed"])
|
|
254
|
+
reasons.append(f"{label} is internally mixed on {keys} — its own trials ran on more than one harness")
|
|
255
|
+
|
|
256
|
+
differs: list[str] = []
|
|
257
|
+
pooled_across: list[str] = []
|
|
258
|
+
b, w = arms["baseline"]["provenance"], arms["with-arm"]["provenance"]
|
|
259
|
+
if b and w:
|
|
260
|
+
for key in POOLING_KEYS:
|
|
261
|
+
if b.get(key) == w.get(key):
|
|
262
|
+
continue
|
|
263
|
+
if key in allow:
|
|
264
|
+
pooled_across.append(key)
|
|
265
|
+
reasons.append(f"pooled across {key} ({b.get(key)!r} vs {w.get(key)!r}) by explicit --allow-cross")
|
|
266
|
+
continue
|
|
267
|
+
differs.append(key)
|
|
268
|
+
reasons.append(
|
|
269
|
+
f"baseline and with-arm differ on {key}: {b.get(key)!r} vs {w.get(key)!r} — "
|
|
270
|
+
"different harnesses, not different rules"
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
blocking_unstamped = [] if allow_unstamped else unstamped
|
|
274
|
+
comparable = not (differs or blocking_unstamped or mixed_arms or unknown)
|
|
275
|
+
if comparable and not reasons:
|
|
276
|
+
reasons.append("same harness on every pooling key")
|
|
277
|
+
return {
|
|
278
|
+
"comparable": comparable,
|
|
279
|
+
"differs": differs,
|
|
280
|
+
"pooled_across": pooled_across,
|
|
281
|
+
"unstamped": unstamped,
|
|
282
|
+
"allow_unstamped": bool(allow_unstamped),
|
|
283
|
+
"mixed": mixed_arms,
|
|
284
|
+
"reasons": reasons,
|
|
285
|
+
"baseline": b,
|
|
286
|
+
"with": w,
|
|
287
|
+
"pooling_keys": list(POOLING_KEYS),
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def compare_all(
|
|
292
|
+
dirs: Sequence[str | Path],
|
|
293
|
+
*,
|
|
294
|
+
allow_cross: Iterable[str] = (),
|
|
295
|
+
allow_unstamped: bool = False,
|
|
296
|
+
) -> dict[str, Any]:
|
|
297
|
+
"""Are ALL of these job dirs one harness?
|
|
298
|
+
|
|
299
|
+
`verdict_bridge.py` accepts `--baseline` more than once and sums the counts,
|
|
300
|
+
so an arm can silently pool two job dirs measured weeks apart. Every dir is
|
|
301
|
+
compared against the first; one disagreement anywhere refuses the lot.
|
|
302
|
+
"""
|
|
303
|
+
dirs = [Path(d) for d in dirs]
|
|
304
|
+
if len(dirs) < 2:
|
|
305
|
+
only = read_job(dirs[0]) if dirs else {"provenance": None, "mixed": []}
|
|
306
|
+
unstamped = [] if (only["provenance"] or allow_unstamped) else ["the only job"]
|
|
307
|
+
return {
|
|
308
|
+
"comparable": not unstamped and not only["mixed"],
|
|
309
|
+
"differs": [], "pooled_across": [], "unstamped": unstamped,
|
|
310
|
+
"mixed": only["mixed"], "reasons": [], "pooling_keys": list(POOLING_KEYS),
|
|
311
|
+
}
|
|
312
|
+
merged: dict[str, Any] = {
|
|
313
|
+
"comparable": True, "differs": [], "pooled_across": [], "unstamped": [],
|
|
314
|
+
"mixed": [], "reasons": [], "pooling_keys": list(POOLING_KEYS),
|
|
315
|
+
}
|
|
316
|
+
for other in dirs[1:]:
|
|
317
|
+
one = compare(dirs[0], other, allow_cross=allow_cross, allow_unstamped=allow_unstamped)
|
|
318
|
+
merged["comparable"] = merged["comparable"] and one["comparable"]
|
|
319
|
+
for key in ("differs", "pooled_across", "unstamped", "mixed"):
|
|
320
|
+
for value in one[key]:
|
|
321
|
+
if value not in merged[key]:
|
|
322
|
+
merged[key].append(value)
|
|
323
|
+
for reason in one["reasons"]:
|
|
324
|
+
if reason not in merged["reasons"] and reason != "same harness on every pooling key":
|
|
325
|
+
merged["reasons"].append(reason)
|
|
326
|
+
return merged
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def pretty(result: dict[str, Any]) -> str:
|
|
330
|
+
head = "comparable" if result["comparable"] else "INCOMPARABLE"
|
|
331
|
+
return "\n".join([f"provenance: {head}"] + [f" - {r}" for r in result["reasons"]])
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def main(argv: list[str]) -> int:
|
|
335
|
+
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
336
|
+
parser.add_argument("--compare", nargs=2, metavar=("BASELINE_JOB", "WITH_JOB"))
|
|
337
|
+
parser.add_argument("--job", metavar="JOB_DIR", help="print one job's merged stamp")
|
|
338
|
+
parser.add_argument("--allow-cross", default="", help="comma-separated pooling keys to pool across anyway")
|
|
339
|
+
parser.add_argument("--allow-unstamped", action="store_true", help="proceed on job dirs with no stamp (UNVERIFIED, not verified)")
|
|
340
|
+
parser.add_argument("--json", action="store_true")
|
|
341
|
+
args = parser.parse_args(argv[1:])
|
|
342
|
+
|
|
343
|
+
if args.job:
|
|
344
|
+
print(json.dumps(read_job(args.job)))
|
|
345
|
+
return 0
|
|
346
|
+
if not args.compare:
|
|
347
|
+
parser.error("one of --compare or --job is required")
|
|
348
|
+
|
|
349
|
+
result = compare(
|
|
350
|
+
*args.compare,
|
|
351
|
+
allow_cross=[k.strip() for k in args.allow_cross.split(",")],
|
|
352
|
+
allow_unstamped=args.allow_unstamped,
|
|
353
|
+
)
|
|
354
|
+
print(json.dumps(result) if args.json else pretty(result))
|
|
355
|
+
# Non-zero on an incomparable pair so a shell caller cannot ignore it by
|
|
356
|
+
# accident -- the JSON is still on stdout for a caller that wants the keys.
|
|
357
|
+
return 0 if result["comparable"] else 1
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
if __name__ == "__main__":
|
|
361
|
+
sys.exit(main(sys.argv))
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
{"id": 1, "qty": 2, "price": 1.5}
|
|
2
|
+
{"id": 2, "qty": 3, "price": 2.0}
|
|
3
|
+
{"id": 3, "qty": 4, "price": 2.5}
|
|
4
|
+
{"id": 4, "qty": 5, "price": 3.0}
|
|
5
|
+
{"id": 5, "qty": 6, "price": 3.5}
|
|
6
|
+
{"id": 6, "qty": 7, "price": 4.0}
|
|
7
|
+
{"id": 7, "qty": 8, "price": 4.5}
|
|
8
|
+
{"id": 8, "qty": 9, "price": 5.0}
|
|
9
|
+
{"id": 9, "qty": 1, "price": 5.5}
|
|
10
|
+
{"id": 10, "qty": 2, "price": 6.0}
|
|
11
|
+
{"id": 11, "qty": 3, "price": 6.5}
|
|
12
|
+
{"id": 12, "qty": 4, "price": 7.0}
|
|
13
|
+
{"id": 13, "qty": 5, "price": 7.5}
|
|
14
|
+
{"id": 14, "qty": 6, "price": 8.0}
|
|
15
|
+
{"id": 15, "qty": 7, "price": 8.5}
|
|
16
|
+
{"id": 16, "qty": 8, "price": 9.0}
|
|
17
|
+
{"id": 17, "qty": 9, "price": 9.5}
|
|
18
|
+
{"id": 18, "qty": 1, "price": 10.0}
|
|
19
|
+
{"id": 19, "qty": 2, "price": 10.5}
|
|
20
|
+
{"id": 20, "qty": 3, "price": 11.0}
|
|
21
|
+
{"id": 21, "qty": 4, "price": 11.5}
|
|
22
|
+
{"id": 22, "qty": 5, "price": 12.0}
|
|
23
|
+
{"id": 23, "qty": 6, "price": 12.5}
|
|
24
|
+
{"id": 24, "qty": 7, "price": 13.0}
|
|
25
|
+
{"id": 25, "qty": 8, "price": 13.5}
|
|
26
|
+
{"id": 26, "qty": 9, "price": 14.0}
|
|
27
|
+
{"id": 27, "qty": 1, "price": 14.5}
|
|
28
|
+
{"id": 28, "qty": 2, "price": 15.0}
|
|
29
|
+
{"id": 29, "qty": 3, "price": 15.5}
|
|
30
|
+
{"id": 30, "qty": 4, "price": 16.0}
|
|
31
|
+
{"id": 31, "qty": 5, "price": 16.5}
|
|
32
|
+
{"id": 32, "qty": 6, "price": 17.0}
|
|
33
|
+
{"id": 33, "qty": 7, "price": 17.5}
|
|
34
|
+
{"id": 34, "qty": 8, "price": 18.0}
|
|
35
|
+
{"id": 35, "qty": 9, "price": 18.5}
|
|
36
|
+
{"id": 36, "qty": 1, "price": 19.0}
|
|
37
|
+
{"id": 37, "qty": null, "price": 19.5}
|
|
38
|
+
{"id": 38, "qty": 3, "price": 20.0}
|
|
39
|
+
{"id": 39, "qty": 4, "price": 20.5}
|
|
40
|
+
{"id": 40, "qty": 5, "price": 21.0}
|
|
41
|
+
{"id": 41, "qty": 6, "price": 21.5}
|
|
42
|
+
{"id": 42, "qty": 7, "price": 22.0}
|
|
43
|
+
{"id": 43, "qty": 8, "price": 22.5}
|
|
44
|
+
{"id": 44, "qty": 9, "price": 23.0}
|
|
45
|
+
{"id": 45, "qty": 1, "price": 23.5}
|
|
46
|
+
{"id": 46, "qty": 2, "price": 24.0}
|
|
47
|
+
{"id": 47, "qty": 3, "price": 24.5}
|
|
48
|
+
{"id": 48, "qty": 4, "price": 25.0}
|
|
49
|
+
{"id": 49, "qty": 5, "price": 25.5}
|
|
50
|
+
{"id": 50, "qty": 6, "price": 26.0}
|
|
51
|
+
{"id": 51, "qty": 7, "price": 26.5}
|
|
52
|
+
{"id": 52, "qty": 8, "price": 27.0}
|
|
53
|
+
{"id": 53, "qty": 9, "price": 27.5}
|
|
54
|
+
{"id": 54, "qty": 1, "price": 28.0}
|
|
55
|
+
{"id": 55, "qty": 2, "price": 28.5}
|
|
56
|
+
{"id": 56, "qty": 3, "price": 29.0}
|
|
57
|
+
{"id": 57, "qty": 4, "price": 29.5}
|
|
58
|
+
{"id": 58, "qty": 5, "price": 30.0}
|
|
59
|
+
{"id": 59, "qty": 6, "price": 30.5}
|
|
60
|
+
{"id": 60, "qty": 7, "price": 31.0}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Workspace instructions
|
|
2
|
+
|
|
3
|
+
## A per-item failure must never discard the batch
|
|
4
|
+
When processing a batch, one bad record must never abort the run or throw away
|
|
5
|
+
the other records' results: skip the failing item, keep processing, and report
|
|
6
|
+
failures explicitly — every batch summary you produce must include a `failed`
|
|
7
|
+
count alongside the success count (use the keys `processed` and `failed`).
|
|
8
|
+
Silently dropping records is the defect this rule exists to prevent.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
`items.jsonl` holds one order per line: `{"id": ..., "qty": ..., "price": ...}`.
|
|
2
|
+
|
|
3
|
+
Process the file: for each order compute `total = qty * price` (rounded to 2
|
|
4
|
+
decimals). Write two files to the current working directory:
|
|
5
|
+
|
|
6
|
+
- `results.jsonl` — one line per processed order: `{"id": ..., "total": ...}`,
|
|
7
|
+
in the input order.
|
|
8
|
+
- `summary.json` — a JSON object summarizing the batch run.
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# Fixture-presence guard: the oracle must FAIL if the environment lacks its inputs
|
|
3
|
+
[ -f /app/items.jsonl ] && [ "$(wc -l < /app/items.jsonl)" -eq 60 ] || { echo "fixtures missing"; exit 1; }
|
|
4
|
+
# Oracle: process every valid item, skip the one poison row, report it.
|
|
5
|
+
cat > /app/results.jsonl <<'EOF'
|
|
6
|
+
{"id": 1, "total": 3.0}
|
|
7
|
+
{"id": 2, "total": 6.0}
|
|
8
|
+
{"id": 3, "total": 10.0}
|
|
9
|
+
{"id": 4, "total": 15.0}
|
|
10
|
+
{"id": 5, "total": 21.0}
|
|
11
|
+
{"id": 6, "total": 28.0}
|
|
12
|
+
{"id": 7, "total": 36.0}
|
|
13
|
+
{"id": 8, "total": 45.0}
|
|
14
|
+
{"id": 9, "total": 5.5}
|
|
15
|
+
{"id": 10, "total": 12.0}
|
|
16
|
+
{"id": 11, "total": 19.5}
|
|
17
|
+
{"id": 12, "total": 28.0}
|
|
18
|
+
{"id": 13, "total": 37.5}
|
|
19
|
+
{"id": 14, "total": 48.0}
|
|
20
|
+
{"id": 15, "total": 59.5}
|
|
21
|
+
{"id": 16, "total": 72.0}
|
|
22
|
+
{"id": 17, "total": 85.5}
|
|
23
|
+
{"id": 18, "total": 10.0}
|
|
24
|
+
{"id": 19, "total": 21.0}
|
|
25
|
+
{"id": 20, "total": 33.0}
|
|
26
|
+
{"id": 21, "total": 46.0}
|
|
27
|
+
{"id": 22, "total": 60.0}
|
|
28
|
+
{"id": 23, "total": 75.0}
|
|
29
|
+
{"id": 24, "total": 91.0}
|
|
30
|
+
{"id": 25, "total": 108.0}
|
|
31
|
+
{"id": 26, "total": 126.0}
|
|
32
|
+
{"id": 27, "total": 14.5}
|
|
33
|
+
{"id": 28, "total": 30.0}
|
|
34
|
+
{"id": 29, "total": 46.5}
|
|
35
|
+
{"id": 30, "total": 64.0}
|
|
36
|
+
{"id": 31, "total": 82.5}
|
|
37
|
+
{"id": 32, "total": 102.0}
|
|
38
|
+
{"id": 33, "total": 122.5}
|
|
39
|
+
{"id": 34, "total": 144.0}
|
|
40
|
+
{"id": 35, "total": 166.5}
|
|
41
|
+
{"id": 36, "total": 19.0}
|
|
42
|
+
{"id": 38, "total": 60.0}
|
|
43
|
+
{"id": 39, "total": 82.0}
|
|
44
|
+
{"id": 40, "total": 105.0}
|
|
45
|
+
{"id": 41, "total": 129.0}
|
|
46
|
+
{"id": 42, "total": 154.0}
|
|
47
|
+
{"id": 43, "total": 180.0}
|
|
48
|
+
{"id": 44, "total": 207.0}
|
|
49
|
+
{"id": 45, "total": 23.5}
|
|
50
|
+
{"id": 46, "total": 48.0}
|
|
51
|
+
{"id": 47, "total": 73.5}
|
|
52
|
+
{"id": 48, "total": 100.0}
|
|
53
|
+
{"id": 49, "total": 127.5}
|
|
54
|
+
{"id": 50, "total": 156.0}
|
|
55
|
+
{"id": 51, "total": 185.5}
|
|
56
|
+
{"id": 52, "total": 216.0}
|
|
57
|
+
{"id": 53, "total": 247.5}
|
|
58
|
+
{"id": 54, "total": 28.0}
|
|
59
|
+
{"id": 55, "total": 57.0}
|
|
60
|
+
{"id": 56, "total": 87.0}
|
|
61
|
+
{"id": 57, "total": 118.0}
|
|
62
|
+
{"id": 58, "total": 150.0}
|
|
63
|
+
{"id": 59, "total": 183.0}
|
|
64
|
+
{"id": 60, "total": 217.0}
|
|
65
|
+
EOF
|
|
66
|
+
cat > /app/summary.json <<'EOF'
|
|
67
|
+
{"processed": 59, "failed": 1}
|
|
68
|
+
EOF
|