@chrono-meta/fh-gate 3.0.0 → 3.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/regression/probes_live.yaml +137 -0
- package/.claude/rules/.residency-patterns.defaults +7 -0
- package/.claude/rules/fh_4axis_gate.md +50 -1
- package/.claude-plugin/marketplace.json +8 -2
- package/AGENTS.md +27 -0
- package/CATALOG.md +17 -0
- package/CLAUDE.md +12 -2
- package/README.ja.md +51 -7
- package/README.ko.md +48 -7
- package/README.md +37 -5
- package/README.zh.md +45 -8
- package/docs/STANDARDS_ALIGNMENT.md +61 -0
- package/docs/USER_GUIDE.md +3 -0
- package/docs/USE_CASES.md +50 -0
- package/docs/model_tier_expectations.md +60 -0
- package/knowledge/shared/harness-core/fh_three_layer_canon.md +30 -10
- package/knowledge/shared/harness-core/field_verdict_crossfamily_gate.md +8 -0
- package/knowledge/shared/harness-core/harness_incubator_doctrine.md +9 -0
- package/knowledge/shared/harness-core/iso_ai_standards_crosswalk.md +139 -0
- package/knowledge/shared/harness-core/measurement-integrity-checklist.md +20 -0
- package/knowledge/shared/learnings/subagent_invocations_log.yaml +451 -4
- package/package.json +40 -2
- package/plugins/fh-commons/.claude-plugin/plugin.json +1 -1
- package/plugins/fh-commons/skills/preprep/README.md +4 -1
- package/plugins/fh-commons/skills/preprep/SKILL.md +99 -3
- package/plugins/fh-commons/skills/preprep/diagram_from_json.py +154 -0
- package/plugins/fh-commons/skills/preprep/fixtures/fixture_R3_negative.pptx +0 -0
- package/plugins/fh-commons/skills/preprep/fixtures/fixture_R3_positive.pptx +0 -0
- package/plugins/fh-commons/skills/preprep/fixtures/mk_slide_fixtures.py +179 -0
- package/plugins/fh-commons/skills/preprep/interslide_deps.py +98 -9
- package/plugins/fh-commons/skills/preprep/lane_adjacent_dup.py +4 -1
- package/plugins/fh-commons/skills/preprep/lane_diagram.py +95 -0
- package/plugins/fh-commons/skills/preprep/lane_geometry.py +181 -0
- package/plugins/fh-commons/skills/preprep/lane_promise.py +4 -1
- package/plugins/fh-commons/skills/preprep/lane_slide_refs.py +134 -0
- package/plugins/fh-commons/skills/preprep/lane_slide_relations.py +300 -0
- package/plugins/fh-commons/skills/preprep/preprep.py +129 -8
- package/plugins/fh-commons/skills/preprep/preprep_wire.py +210 -0
- package/plugins/fh-commons/skills/preprep/presentation_checklist.md +4 -0
- package/plugins/fh-commons/skills/preprep/surfaces.example.yaml +18 -0
- package/plugins/fh-commons/skills/preprep/test_preprep_lanes_rp.py +184 -0
- package/plugins/fh-meta/.claude-plugin/plugin.json +1 -1
- package/plugins/fh-meta/CHANGELOG.md +76 -1
- package/plugins/fh-meta/skills/auto-decorrelation/SKILL.md +87 -2
- package/plugins/fh-meta/skills/frontier-digest/SKILL.md +2 -2
- package/plugins/fh-meta/skills/frontier-digest/SKILL_detail.md +25 -5
- package/plugins/fh-meta/skills/hub-cc-pr-reviewer/SKILL.md +11 -0
- package/plugins/fh-meta/skills/hub-cc-pr-reviewer/SKILL_detail.md +3 -0
- package/plugins/fh-qp/.claude-plugin/plugin.json +22 -0
- package/plugins/fh-qp/README.md +71 -0
- package/plugins/fh-qp/fixtures/evidence_known_clean.txt +3 -0
- package/plugins/fh-qp/fixtures/evidence_known_dirty.txt +6 -0
- package/plugins/fh-qp/fixtures/reach_known_partial.tsv +4 -0
- package/plugins/fh-qp/fixtures/reach_known_reached.tsv +4 -0
- package/plugins/fh-qp/fixtures/reach_known_wall.tsv +3 -0
- package/plugins/fh-qp/fixtures/verdicts_known_bad_branch.tsv +2 -0
- package/plugins/fh-qp/fixtures/verdicts_known_bad_vacuous_machine.tsv +2 -0
- package/plugins/fh-qp/fixtures/verdicts_known_good.tsv +4 -0
- package/plugins/fh-qp/fixtures/verdicts_verify_only.tsv +3 -0
- package/plugins/fh-qp/qp_profile.example.yaml +29 -0
- package/plugins/fh-qp/scripts/qp_tools.sh +216 -0
- package/plugins/fh-qp/skills/qp/SKILL.md +65 -0
- package/plugins/fh-qp/skills/qp-plan/SKILL.md +42 -0
- package/plugins/fh-qp/skills/qp-regress/SKILL.md +45 -0
- package/plugins/fh-qp/skills/qp-run/SKILL.md +49 -0
- package/scripts/chamber_run.sh +14 -5
- package/scripts/com.forge-harness.live-eval.plist +84 -0
- package/scripts/compaction_probe.sh +9 -35
- package/scripts/directional_diff_gate.sh +14 -2
- package/scripts/frontier_digest_autopilot.sh +4 -1
- package/scripts/map_postprocess.py +90 -0
- package/scripts/outbound_query_guard.sh +131 -0
- package/scripts/outbound_query_hook.sh +373 -0
- package/scripts/package_coverage_check.sh +55 -16
- package/scripts/pipe_verdict_guard.sh +41 -1
- package/scripts/probe_live_eval.sh +240 -0
- package/scripts/probe_live_eval_lib.py +579 -0
- package/scripts/proposal_hook.sh +120 -17
- package/scripts/push_zone_check.sh +78 -0
- package/scripts/residency_closure_scan.py +252 -0
- package/scripts/selfcheck.sh +41 -1
- package/scripts/session_close_check.sh +100 -0
- package/scripts/sim_isolated_run.sh +98 -2
- package/scripts/test_action_yml_lanes.sh +160 -0
- package/scripts/test_fh_qp_lanes.sh +105 -0
- package/scripts/test_gate_two_verdicts_lanes.sh +139 -0
- package/scripts/test_map_postprocess_lanes.sh +143 -0
- package/scripts/test_marker_affected_lanes.sh +93 -0
- package/scripts/test_marker_crossfamily_lanes.sh +90 -6
- package/scripts/test_marker_oracle_lanes.sh +136 -0
- package/scripts/test_outbound_query_hook_lanes.sh +433 -0
- package/scripts/test_outbound_query_lanes.sh +87 -0
- package/scripts/test_pipe_verdict_guard_lanes.sh +26 -0
- package/scripts/test_preprep_diagram_lanes.sh +87 -0
- package/scripts/test_preprep_drift_anchor.sh +3 -3
- package/scripts/test_preprep_slide_refs_lanes.sh +169 -0
- package/scripts/test_probe_live_eval_lanes.sh +437 -0
- package/scripts/test_proposal_hook_lanes.sh +22 -1
- package/scripts/test_push_zone_lanes.sh +304 -0
- package/scripts/test_residency_closure_lanes.sh +70 -0
- package/scripts/test_sim_isolated_run_lanes.sh +119 -0
- package/scripts/test_utterance_intake_lanes.sh +414 -0
- package/scripts/test_worktree_reclaim_lanes.sh +70 -0
- package/scripts/transcript_utterances.py +222 -0
- package/scripts/utterance_intake.sh +424 -0
- package/scripts/validate_yaml.sh +27 -0
- package/scripts/worktree_reclaim.sh +95 -0
- package/templates/.git-hooks/pre-commit +353 -1
- package/templates/.git-hooks/pre-push +91 -0
- package/templates/RED_TEAM_REPORT.md +49 -0
- package/templates/settings.PreToolUse.snippet.json +65 -1
|
@@ -138,14 +138,54 @@ if ! printf '%s' "$NORM" | grep -qE 'set -o pipefail|set -[a-zA-Z]*o[a-zA-Z]* pi
|
|
|
138
138
|
# Branch (b)'s optional `([^|&]*;)?` prefix keeps `| (sort; tail -5); echo $?` caught — the
|
|
139
139
|
# filter need not be the group's FIRST command, only its LAST — while the `;` requirement keeps
|
|
140
140
|
# the filter at a statement start, so `| (echo cat); rc=$?` does not match through a bare word.
|
|
141
|
+
# 🟥 «바로 다음 문장» 으로 좁힌다 (2026-09-06, 운영자 실사용 신고 + known-pair 재현).
|
|
142
|
+
# 종전 꼬리는 `[;&].*\$\?` 라 파이프 뒤 **어디든** `$?` 가 있으면 잡았다 — 그게 **다른
|
|
143
|
+
# 명령의** `$?` 여도. 한 호출에 문장이 여럿인 실사용 커맨드는 거의 항상 걸린다.
|
|
144
|
+
# 실측 4팔: ①`out=$(c); rc=$?; echo "$out"|tail` 안 터짐(정상) ②`c|tail; echo $?` 터짐
|
|
145
|
+
# (known-positive) ③표시전용 안 터짐 ④`out=$(a); rc=$?; echo "$out"|tail; b; echo $?`
|
|
146
|
+
# **터짐 = 오탐**. ④의 `$?` 는 `b` 의 것이라 안 잡는 것이 옳다.
|
|
147
|
+
# ⇒ 꼬리를 `[;&]&?[^;&|]*\$\?` 로 — 파이프라인 **직후 한 문장** 안에서만 읽는다.
|
|
148
|
+
# `&?` 는 `cmd | tail && echo $?`(진짜 결함)를 살리기 위한 것이다.
|
|
149
|
+
# 이 좁힘의 재현율 손실은 사실상 0 이다: 사이에 명령이 끼면 `$?` 는 그 명령의 것이므로
|
|
150
|
+
# 원래 **잡으면 안 되는** 자리다. 훅 자기 주석이 «100% FP 는 정작 중요한 한 건을
|
|
151
|
+
# 무시하도록 훈련시킨다» 고 적었고, 운영자 신고가 정확히 그 신호였다.
|
|
141
152
|
_F='(tail|head|cat|less|more)'
|
|
142
153
|
if printf '%s' "$NORM" \
|
|
143
|
-
| grep -qE "\|&?[[:space:]]*(${_F}([[:space:]][^|;&]*)?|[({]([^|&]*;)?[[:space:]]*${_F}([[:space:]][^|;&()}]*)?[[:space:]]*;?[[:space:]]*[)}])[[:space:]]*[;&]
|
|
154
|
+
| grep -qE "\|&?[[:space:]]*(${_F}([[:space:]][^|;&]*)?|[({]([^|&]*;)?[[:space:]]*${_F}([[:space:]][^|;&()}]*)?[[:space:]]*;?[[:space:]]*[)}])[[:space:]]*[;&]&?[^;&|]*\\\$\?"; then
|
|
144
155
|
add "R2 \$? after a display filter" \
|
|
145
156
|
"\$? holds the filter's status (tail/head/cat almost always succeed), not the command's — a FAILED check reads as 0. Capture first: \`out=\$(cmd 2>&1); rc=\$?\` then print \"\$out\" | tail."
|
|
146
157
|
fi
|
|
147
158
|
fi
|
|
148
159
|
|
|
160
|
+
# ── R3 — a line that is ONLY redirections (`2>&1` alone on the line after a heredoc). ─────────
|
|
161
|
+
# zsh runs a redirection-only line as `$NULLCMD` (= `cat` by default) — measured 2026-09-04 with
|
|
162
|
+
# `zsh -x`: `+zsh:1> cat`. That `cat` reads STDIN until EOF. In this Bash tool, stdin is `/dev/null`
|
|
163
|
+
# ONLY when the harness appends `< /dev/null`, and it does NOT append it when the top-level command
|
|
164
|
+
# already carries a stdin redirect (`<` or a `<<` heredoc) — then stdin is a pipe held open for the
|
|
165
|
+
# life of the command, and `cat` never returns. Measured 2026-09-04: 9 tasks in two sessions, all of
|
|
166
|
+
# the shape `out=$(git commit -q -F - <<'CEOF' … CEOF ⏎ 2>&1)`: every commit landed within 4 s, no
|
|
167
|
+
# push ever started, the shell sat in `$( )` with no git child, killed later (exit 144). The
|
|
168
|
+
# pre-push hook never ran — it was blamed for a hang that happened before it.
|
|
169
|
+
# Heredoc BODIES are stripped first: a one-word markdown blockquote line (`> 정본`) inside a commit
|
|
170
|
+
# message is data, not a redirection. Deterministic on the remaining lines; zero-FP by construction
|
|
171
|
+
# (a redirection-only line is never what the author meant outside a heredoc body).
|
|
172
|
+
_R3_HIT=$(printf '%s\n' "$CMD" | awk '
|
|
173
|
+
BEGIN { inhd = 0; d = "" }
|
|
174
|
+
inhd { if ($0 == d || $0 ~ ("^[[:space:]]*" d "$")) { inhd = 0 }; next }
|
|
175
|
+
{
|
|
176
|
+
if (match($0, /<<-?[[:space:]]*["'"'"']?[A-Za-z_][A-Za-z_0-9]*["'"'"']?/)) {
|
|
177
|
+
d = substr($0, RSTART, RLENGTH); sub(/^<<-?[[:space:]]*/, "", d); gsub(/["'"'"']/, "", d); inhd = 1
|
|
178
|
+
}
|
|
179
|
+
# a STATEMENT at line start made only of redirections, ended by `)`, `;`, `}`, `&&`, `||` or EOL —
|
|
180
|
+
# the measured shape is `2>&1); rc=$?`, where `)` closes the `$( )` and the last statement inside
|
|
181
|
+
# it is the bare `2>&1`.
|
|
182
|
+
if ($0 ~ /^[[:space:]]*[0-9]*(>>?|<)(&[0-9]+|[[:space:]]*[^[:space:];|&<>()]+)?([[:space:]]+[0-9]*(>>?|<)(&[0-9]+|[[:space:]]*[^[:space:];|&<>()]+)?)*[[:space:]]*([;)}]|&&|\|\||$)/) { print NR ": " $0; exit }
|
|
183
|
+
}')
|
|
184
|
+
if [ -n "$_R3_HIT" ]; then
|
|
185
|
+
add "R3 redirection-only line runs \`cat\` on stdin (line ${_R3_HIT})" \
|
|
186
|
+
"zsh executes a line that is only redirections as \$NULLCMD=cat, which reads stdin to EOF. This command carries a top-level \`<\`/\`<<\`, so the harness leaves stdin as an OPEN PIPE — that cat never returns and everything after it (the push) never starts. Put the redirection on the command line: \`out=\$(git commit -q -F - 2>&1 <<'"'"'CEOF'"'"'\`."
|
|
187
|
+
fi
|
|
188
|
+
|
|
149
189
|
[ -n "$hits" ] || exit 0
|
|
150
190
|
|
|
151
191
|
if [ "${FH_PIPE_VERDICT_BLOCK:-0}" = "1" ]; then
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# probe_live_eval.sh — the LIVE twin of /prompt-regression's static probe check.
|
|
3
|
+
#
|
|
4
|
+
# WHY THIS EXISTS. `plugins/fh-meta/skills/prompt-regression/SKILL.md` says this about itself,
|
|
5
|
+
# verbatim, in its own §Step 4: "What it therefore cannot catch — a rule that is still present but
|
|
6
|
+
# has stopped FIRING (salience loss, ordering, competition from another rule) — a trigger phrase
|
|
7
|
+
# present in the file but shadowed by a higher-priority route — any behavior change that leaves the
|
|
8
|
+
# source text identical." That gap is exactly what an eval-style harness closes (CLAUDE.md
|
|
9
|
+
# §Autonomous Initiative Layer cites the shape: settings-changed PR + a scored run against real
|
|
10
|
+
# tasks, not a source grep). This script IS that harness for FH's own golden probe set
|
|
11
|
+
# (.claude/regression/probes.md): it sends each selected probe's literal utterance to a floor-tier
|
|
12
|
+
# `claude -p` inside an ISOLATED, disposable clone (via scripts/sim_isolated_run.sh — never the live
|
|
13
|
+
# repo, per that script's own header) and greps the ACTUAL RESPONSE, not the source.
|
|
14
|
+
#
|
|
15
|
+
# WHAT IT IS NOT. It does not replace /prompt-regression (source-correctness stays cheap and
|
|
16
|
+
# instant) and it does not replace a blind sim-conductor persona run (which reads a whole artifact
|
|
17
|
+
# for open-ended judgment). This is narrow, mechanical, known-answer, and cheap enough to run
|
|
18
|
+
# nightly — the same trade CLAUDE.md's Measured-Loop memory entry describes for any recurring
|
|
19
|
+
# measurement: sealed pre-registration (probes_live.yaml is authored BEFORE a run, not fit to one),
|
|
20
|
+
# one variable per probe (ARM=primary utterance, CTRL=known-negative), a scorer that runs before
|
|
21
|
+
# results are read, and the falsification condition (a probe whose control does not discriminate
|
|
22
|
+
# is not scored PASS/FAIL — see §Scoring) executed exactly as it is written.
|
|
23
|
+
#
|
|
24
|
+
# SELECTION. Not every probes.md row is live-runnable — see .claude/regression/probes_live.yaml's
|
|
25
|
+
# header for the 4-step mechanical rule (class filter, utterance-shape filter, INERT-ANCHOR filter,
|
|
26
|
+
# hand-curated CLI-event filter) and scripts/probe_live_eval_lib.py for the implementation. Run
|
|
27
|
+
# `--dry-run` to see the full 33-row breakdown: which probes are selected, which are excluded and
|
|
28
|
+
# why, and which pass the mechanical rule but have no authored expect/control yet
|
|
29
|
+
# (NOT-YET-AUTHORED — an honest gap, not a silent drop; G-LINT-01 is the current example, deferred
|
|
30
|
+
# because scoring a full /harness-doctor Step 5 run needs more than a keyword regex).
|
|
31
|
+
#
|
|
32
|
+
# COST. Each selected probe costs TWO live `claude -p` calls (primary + control). Do not run the
|
|
33
|
+
# full selected set casually — use --subset N or --ids P1,P2 for a spot-check, and read
|
|
34
|
+
# sim_isolated_run.sh's own header before running unattended (isolation guarantees, what "observe"
|
|
35
|
+
# mode does and does not prevent, the three-valued rc/bytes verdict for a timeout vs an empty
|
|
36
|
+
# answer).
|
|
37
|
+
#
|
|
38
|
+
# 🟥 «미실행 ≠ 0» — a probe whose primary or control call produced 0 bytes (timeout, rate limit,
|
|
39
|
+
# crash) is scored FAILED-TO-RUN, excluded from the pass-rate denominator, and its count is
|
|
40
|
+
# reported separately. A FAILED-TO-RUN probe is not evidence the behavior is absent — it is
|
|
41
|
+
# evidence nothing was measured (CLAUDE.md §Instrument-Calibration: "not found ≠ 0").
|
|
42
|
+
#
|
|
43
|
+
# SCORING (see scripts/probe_live_eval_lib.py:score_probe for the exact rule). Each probe declares
|
|
44
|
+
# a `polarity` in probes_live.yaml:
|
|
45
|
+
# present — expect_re must appear in PRIMARY and must NOT appear in CONTROL
|
|
46
|
+
# absent — expect_re must NOT appear in PRIMARY and MUST appear in CONTROL
|
|
47
|
+
# Either direction of "the control disagrees with what polarity predicts" scores that probe
|
|
48
|
+
# UNCALIBRATED, never PASS or FAIL — an instrument that cannot discriminate has not measured
|
|
49
|
+
# anything, per this repo's own instrument-calibration discipline. If ANY probe in a run is
|
|
50
|
+
# UNCALIBRATED, the WHOLE RUN's overall verdict is UNCALIBRATED (rc=2) — a pass rate computed
|
|
51
|
+
# alongside a proven-blind probe is not trustworthy just because the other rows look fine.
|
|
52
|
+
#
|
|
53
|
+
# EXIT CODES: 0 = PASS (pass_rate >= threshold, no UNCALIBRATED). 1 = FAIL (pass_rate < threshold).
|
|
54
|
+
# 2 = UNCALIBRATED, NO-PROBES-RAN, or a RUNNER PREFLIGHT FAILURE (sim_isolated_run.sh's own usage
|
|
55
|
+
# guards exit 2 before ever calling `claude` — e.g. no `claude` on PATH, or — measured 2026-09-05 —
|
|
56
|
+
# a launchd PATH with no `timeout(1)` on it). This script fails fast on the FIRST such rc=2 rather
|
|
57
|
+
# than burning the remaining clones against an environment already known broken: no verdict
|
|
58
|
+
# rendered either way — fix the instrument before trusting it. See sim_isolated_run.sh's own
|
|
59
|
+
# §timeout(1) RESOLUTION header for why that preflight can fail even when `claude` itself is fine.
|
|
60
|
+
#
|
|
61
|
+
# USAGE
|
|
62
|
+
# bash scripts/probe_live_eval.sh --dry-run
|
|
63
|
+
# bash scripts/probe_live_eval.sh --ids G-GREET-01,G-TRIG-02 --model sonnet
|
|
64
|
+
# bash scripts/probe_live_eval.sh --subset 3
|
|
65
|
+
# bash scripts/probe_live_eval.sh --subset 3 --report-out /tmp/spot.md # spot-check: keep the nightly record untouched
|
|
66
|
+
# bash scripts/probe_live_eval.sh # full selected set — nightly cron shape
|
|
67
|
+
#
|
|
68
|
+
# PORTABILITY: bash 3.2 (macOS) + bash 5.x (Linux CI). No associative arrays, no `${var,,}`,
|
|
69
|
+
# heredocs only inside functions with quoted delimiters (see [[feedback_unquoted_heredoc_backtick_executes]]
|
|
70
|
+
# — this script has none; the one heredoc-shaped block lives in probe_live_eval_lib.py, a real
|
|
71
|
+
# file, not a bash heredoc).
|
|
72
|
+
|
|
73
|
+
set -uo pipefail
|
|
74
|
+
|
|
75
|
+
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
76
|
+
PROBES_MD="$REPO_ROOT/.claude/regression/probes.md"
|
|
77
|
+
PROBES_LIVE="$REPO_ROOT/.claude/regression/probes_live.yaml"
|
|
78
|
+
LIB="$REPO_ROOT/scripts/probe_live_eval_lib.py"
|
|
79
|
+
# FH_SIM_RUNNER_BIN — same override name test_sim_isolated_run_lanes.sh already uses for the same
|
|
80
|
+
# purpose (point at an alternate runner build). Here it also lets test_probe_live_eval_lanes.sh
|
|
81
|
+
# swap in a stub runner (rc=2, no `claude` call, no network) to test the fail-fast behavior below
|
|
82
|
+
# without spawning a live session. No-op when unset — default behavior is unchanged.
|
|
83
|
+
SIM_RUNNER="${FH_SIM_RUNNER_BIN:-$REPO_ROOT/scripts/sim_isolated_run.sh}"
|
|
84
|
+
|
|
85
|
+
# ── file-header constant — the "문턱" the design brief calls for. Change here, not per-invocation. ──
|
|
86
|
+
THRESHOLD="0.8"
|
|
87
|
+
|
|
88
|
+
MODEL="sonnet"
|
|
89
|
+
# ── reps — 프로브당 반복 횟수. 기본 1(종전 행동 그대로), 무인 런은 plist 에서 3 을 준다. ──
|
|
90
|
+
# 🟥 왜 1 이 기본이면서 야간은 3 인가 (2026-09-06 실측): 유효 런 3 개를 재채점하니 12 프로브 중
|
|
91
|
+
# 5 개가 flaky 였고, **같은 코퍼스·15분 간격**의 두 런에서 4 개가 뒤집혔다. 관측 pass_rate 는
|
|
92
|
+
# 0.50 / 0.67 / 0.67 — 단일 rep 의 노이즈 폭이 문턱 0.80 까지의 거리보다 넓다. 즉 reps=1 짜리
|
|
93
|
+
# pass_rate 로는 문턱을 정할 수 없다. 채점은 과반이고, 분산은 리포트의 `reps(pass/ran)` 칸에
|
|
94
|
+
# 그대로 남긴다(3/3 과 2/3 은 다른 사실이다).
|
|
95
|
+
# ⚠️ 비용: reps=3 이면 프로브당 `claude -p` 호출이 2 → 6 이다(12 프로브 = 24 → 72).
|
|
96
|
+
REPS=1
|
|
97
|
+
DRYRUN=0
|
|
98
|
+
SUBSET=""
|
|
99
|
+
IDS=""
|
|
100
|
+
OUTDIR=""
|
|
101
|
+
REPORT_OUT="" # --report-out: where the markdown report lands (default: tracks/_meta/live_eval_<date>.md)
|
|
102
|
+
while [ $# -gt 0 ]; do
|
|
103
|
+
case "$1" in
|
|
104
|
+
--subset) SUBSET="${2:-}"; shift 2 ;;
|
|
105
|
+
--ids) IDS="${2:-}"; shift 2 ;;
|
|
106
|
+
--model) MODEL="${2:-sonnet}"; shift 2 ;;
|
|
107
|
+
--reps) REPS="${2:-1}"; shift 2 ;;
|
|
108
|
+
--dry-run) DRYRUN=1; shift ;;
|
|
109
|
+
--out) OUTDIR="${2:-}"; shift 2 ;;
|
|
110
|
+
--report-out) REPORT_OUT="${2:-}"; shift 2 ;; # lanes/spot-checks MUST pass this — never the live path
|
|
111
|
+
*) echo "unknown flag: $1" >&2; exit 2 ;;
|
|
112
|
+
esac
|
|
113
|
+
done
|
|
114
|
+
|
|
115
|
+
case "$REPS" in ''|*[!0-9]*) echo "FAIL: --reps must be a positive integer (got '$REPS')" >&2; exit 2 ;; esac
|
|
116
|
+
[ "$REPS" -ge 1 ] || { echo "FAIL: --reps must be >= 1 (got '$REPS')" >&2; exit 2; }
|
|
117
|
+
|
|
118
|
+
command -v python3 >/dev/null 2>&1 || { echo "FAIL: python3 required" >&2; exit 2; }
|
|
119
|
+
[ -f "$PROBES_MD" ] || { echo "FAIL: $PROBES_MD not found" >&2; exit 2; }
|
|
120
|
+
[ -f "$PROBES_LIVE" ] || { echo "FAIL: $PROBES_LIVE not found" >&2; exit 2; }
|
|
121
|
+
[ -f "$LIB" ] || { echo "FAIL: $LIB not found" >&2; exit 2; }
|
|
122
|
+
|
|
123
|
+
WORKDIR="$(mktemp -d "${TMPDIR:-/tmp}/fh-live-eval-XXXXXX")"
|
|
124
|
+
SPEC_DIR="$WORKDIR/spec"
|
|
125
|
+
SELECT_JSON="$WORKDIR/select.json"
|
|
126
|
+
|
|
127
|
+
SELECT_ARGS=(select --probes-md "$PROBES_MD" --probes-live "$PROBES_LIVE"
|
|
128
|
+
--json-out "$SELECT_JSON" --spec-dir "$SPEC_DIR")
|
|
129
|
+
[ -n "$SUBSET" ] && SELECT_ARGS+=(--subset "$SUBSET")
|
|
130
|
+
[ -n "$IDS" ] && SELECT_ARGS+=(--ids "$IDS")
|
|
131
|
+
|
|
132
|
+
python3 "$LIB" "${SELECT_ARGS[@]}"
|
|
133
|
+
select_rc=$?
|
|
134
|
+
# select_rc nonzero = a DEAD-POINTER (probes_live.yaml names an id absent from probes.md). That is
|
|
135
|
+
# an authoring bug in this repo's own asset, not a runtime condition — fail loudly rather than
|
|
136
|
+
# silently running a smaller set than intended.
|
|
137
|
+
if [ "$select_rc" -ne 0 ]; then
|
|
138
|
+
echo "" >&2
|
|
139
|
+
echo "❌ selection reported a dead pointer — fix .claude/regression/probes_live.yaml before running." >&2
|
|
140
|
+
rm -rf "$WORKDIR"
|
|
141
|
+
exit "$select_rc"
|
|
142
|
+
fi
|
|
143
|
+
|
|
144
|
+
if [ "$DRYRUN" -eq 1 ]; then
|
|
145
|
+
rm -rf "$WORKDIR"
|
|
146
|
+
exit 0
|
|
147
|
+
fi
|
|
148
|
+
|
|
149
|
+
[ -f "$SPEC_DIR/selected_ids.txt" ] || { echo "FAIL: selection produced no spec dir" >&2; rm -rf "$WORKDIR"; exit 2; }
|
|
150
|
+
SELECTED_COUNT=$(wc -l < "$SPEC_DIR/selected_ids.txt" | tr -d ' ')
|
|
151
|
+
if [ "$SELECTED_COUNT" -eq 0 ]; then
|
|
152
|
+
echo "❌ 0 probes selected after filtering — nothing to run (check --ids / --subset against the" >&2
|
|
153
|
+
echo " SELECTED list printed above)." >&2
|
|
154
|
+
rm -rf "$WORKDIR"
|
|
155
|
+
exit 2
|
|
156
|
+
fi
|
|
157
|
+
|
|
158
|
+
# `claude` presence is the REAL runner's preflight, so it is checked only on that path. Under
|
|
159
|
+
# FH_SIM_RUNNER_BIN the stub owns its own preflight — measured 2026-09-05 on CI (ubuntu, no
|
|
160
|
+
# `claude` installed): this line fired BEFORE the stub was ever called, so FF2/FF3/FF4/FF6 went
|
|
161
|
+
# red while FF1 passed by coincidence (both paths exit 2). Lane FF7 pins the seam; FF7b pins
|
|
162
|
+
# that the real path still refuses to run without `claude`.
|
|
163
|
+
if [ -z "${FH_SIM_RUNNER_BIN:-}" ]; then
|
|
164
|
+
command -v claude >/dev/null 2>&1 || { echo "FAIL: claude CLI not on PATH — cannot run live" >&2; rm -rf "$WORKDIR"; exit 2; }
|
|
165
|
+
fi
|
|
166
|
+
|
|
167
|
+
RUN_DATE="$(date +%Y-%m-%d)"
|
|
168
|
+
OUTDIR="${OUTDIR:-$WORKDIR/run}"
|
|
169
|
+
mkdir -p "$OUTDIR"
|
|
170
|
+
|
|
171
|
+
echo ""
|
|
172
|
+
echo "── live run: $SELECTED_COUNT probe(s), model=$MODEL, out=$OUTDIR ──────────────────────"
|
|
173
|
+
|
|
174
|
+
while IFS= read -r id; do
|
|
175
|
+
[ -z "$id" ] && continue
|
|
176
|
+
input_text="$(cat "$SPEC_DIR/$id.input.txt")"
|
|
177
|
+
control_text="$(cat "$SPEC_DIR/$id.control.txt")"
|
|
178
|
+
probe_out="$OUTDIR/$id"
|
|
179
|
+
mkdir -p "$probe_out"
|
|
180
|
+
|
|
181
|
+
echo ""
|
|
182
|
+
echo "▶ $id — primary"
|
|
183
|
+
bash "$SIM_RUNNER" --arm primary --reps "$REPS" --prompt "$input_text" \
|
|
184
|
+
--mode observe --model "$MODEL" --out "$probe_out" \
|
|
185
|
+
> "$probe_out/_runner_primary.log" 2>&1
|
|
186
|
+
runner_rc=$?
|
|
187
|
+
tail -n 6 "$probe_out/_runner_primary.log"
|
|
188
|
+
# 🟥 fail-fast (2026-09-05) — rc=2 from the runner means its OWN usage/preflight guard tripped
|
|
189
|
+
# before `claude` was ever invoked (missing --arm/--prompt, bogus --mode, no `claude` on PATH,
|
|
190
|
+
# or — the incident this exists for — a launchd PATH with no `timeout(1)` resolvable). That
|
|
191
|
+
# condition is identical for every remaining probe in this run, so continuing would just burn
|
|
192
|
+
# the rest of the clones (up to 2*(N-1) more `claude -p` calls) to the same FAILED-TO-RUN wall.
|
|
193
|
+
# Abort loudly instead of quietly producing a 12/12 FAILED-TO-RUN report with no clue why.
|
|
194
|
+
if [ "$runner_rc" -eq 2 ]; then
|
|
195
|
+
echo "" >&2
|
|
196
|
+
echo "❌ $id primary runner call exited 2 (preflight failure, before \`claude\` ran)." >&2
|
|
197
|
+
echo " Aborting the whole run rather than burning the remaining clones." >&2
|
|
198
|
+
echo " Runner log: $probe_out/_runner_primary.log" >&2
|
|
199
|
+
echo " Partial run artifacts kept at: $OUTDIR" >&2
|
|
200
|
+
exit 2
|
|
201
|
+
fi
|
|
202
|
+
|
|
203
|
+
echo "▶ $id — control"
|
|
204
|
+
bash "$SIM_RUNNER" --arm control --reps "$REPS" --prompt "$control_text" \
|
|
205
|
+
--mode observe --model "$MODEL" --out "$probe_out" \
|
|
206
|
+
> "$probe_out/_runner_control.log" 2>&1
|
|
207
|
+
runner_rc=$?
|
|
208
|
+
tail -n 6 "$probe_out/_runner_control.log"
|
|
209
|
+
if [ "$runner_rc" -eq 2 ]; then
|
|
210
|
+
echo "" >&2
|
|
211
|
+
echo "❌ $id control runner call exited 2 (preflight failure, before \`claude\` ran)." >&2
|
|
212
|
+
echo " Aborting the whole run rather than burning the remaining clones." >&2
|
|
213
|
+
echo " Runner log: $probe_out/_runner_control.log" >&2
|
|
214
|
+
echo " Partial run artifacts kept at: $OUTDIR" >&2
|
|
215
|
+
exit 2
|
|
216
|
+
fi
|
|
217
|
+
done < "$SPEC_DIR/selected_ids.txt"
|
|
218
|
+
|
|
219
|
+
echo ""
|
|
220
|
+
# 🟥 The live report path is the nightly RECORD. A lane or spot-check that reaches this line with the
|
|
221
|
+
# default overwrote a real night's distribution once (2026-09-05 10:18: test_probe_live_eval_lanes.sh FF4
|
|
222
|
+
# ran the real script with a stub runner and replaced the 02:30 report). Lanes pass --report-out.
|
|
223
|
+
REPORT_PATH="${REPORT_OUT:-$REPO_ROOT/tracks/_meta/live_eval_${RUN_DATE}.md}"
|
|
224
|
+
python3 "$LIB" score \
|
|
225
|
+
--probes-live "$PROBES_LIVE" \
|
|
226
|
+
--select-json "$SELECT_JSON" \
|
|
227
|
+
--ids-file "$SPEC_DIR/selected_ids.txt" \
|
|
228
|
+
--run-root "$OUTDIR" \
|
|
229
|
+
--threshold "$THRESHOLD" \
|
|
230
|
+
--model "$MODEL" \
|
|
231
|
+
--report-out "$REPORT_PATH" \
|
|
232
|
+
--run-date "$RUN_DATE" \
|
|
233
|
+
--reps "$REPS"
|
|
234
|
+
score_rc=$?
|
|
235
|
+
|
|
236
|
+
echo ""
|
|
237
|
+
echo "run artifacts kept at: $OUTDIR"
|
|
238
|
+
echo "(temp selection workspace $WORKDIR is NOT auto-deleted when --out was passed explicitly; ok to remove by hand)"
|
|
239
|
+
|
|
240
|
+
exit "$score_rc"
|