@chrono-meta/fh-gate 1.4.88 → 1.4.90
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/judgment_circuits.txt +14 -0
- package/.claude/rules/fh_4axis_gate.md +7 -0
- package/.claude-plugin/marketplace.json +2 -2
- package/AGENTS.md +25 -0
- package/CLAUDE.md +215 -12
- package/knowledge/shared/harness-core/claude_md_gate_details.md +37 -0
- package/knowledge/shared/harness-core/dispatch_conditional_prohibition.md +105 -0
- package/knowledge/shared/harness-core/fh_three_layer_canon.md +165 -0
- package/knowledge/shared/harness-core/harness_incubator_doctrine.md +100 -0
- package/knowledge/shared/harness-core/onboarding_acceleration_autopilot.md +3 -1
- package/knowledge/shared/harness-core/ship_readiness_gate.md +181 -13
- package/knowledge/shared/learnings/subagent_invocations_log.yaml +601 -0
- package/package.json +19 -3
- package/plugins/fh-commons/.claude-plugin/plugin.json +1 -1
- package/plugins/fh-meta/.claude-plugin/plugin.json +1 -1
- package/plugins/fh-meta/skills/auto-decorrelation/SKILL.md +56 -8
- package/plugins/fh-meta/skills/install-wizard/SKILL.md +33 -0
- package/plugins/fh-meta/skills/install-wizard/SKILL_detail.md +104 -15
- package/scripts/chamber_run.sh +64 -2
- package/scripts/chamber_witness.sh +439 -0
- package/scripts/compaction_probe.sh +456 -0
- package/scripts/digest_landing_check.sh +385 -0
- package/scripts/directional_diff_gate.sh +459 -0
- package/scripts/judgment_circuit_lint.sh +239 -0
- package/scripts/novelty_claim_check.sh +193 -0
- package/scripts/prepush_guard_check.sh +15 -0
- package/scripts/psa_scan_lib.sh +36 -0
- package/scripts/relay_channel.sh +645 -0
- package/scripts/reviewer_capability_corpus.tsv +124 -0
- package/scripts/selfcheck.sh +204 -17
- package/scripts/session_close_check.sh +55 -5
- package/scripts/test_marker_crossfamily_lanes.sh +132 -0
- package/scripts/test_marker_floor_lanes.sh +9 -8
- package/scripts/test_relay_channel_lanes.sh +583 -0
- package/scripts/test_reviewer_capability_conformance.sh +173 -0
- package/scripts/test_selfcheck_state_lanes.sh +103 -0
- package/scripts/test_session_close_chain_lanes.sh +79 -4
- package/scripts/test_version_lockstep_lanes.sh +82 -0
- package/scripts/test_wizard_snippet_merge_lanes.sh +104 -11
- package/scripts/universal_guard_check.sh +19 -6
- package/scripts/utterance_landing_check.sh +209 -0
- package/scripts/version_lockstep_check.sh +80 -0
- package/templates/.git-hooks/pre-commit +261 -13
- package/templates/.git-hooks/pre-push +14 -2
- package/templates/settings.Compaction.snippet.json +56 -0
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# reviewer_capability_corpus.tsv — SHARED known-pair corpus for "can this model be an
|
|
2
|
+
|
|
3
|
+
# adversarial reviewer?". Language-neutral DATA, deliberately not code.
|
|
4
|
+
#
|
|
5
|
+
# WHY DATA AND NOT A SHARED LIBRARY. The same judgment lives in at least three places, in
|
|
6
|
+
# three languages, across a public/private repo boundary: a public gate hook (bash), a
|
|
7
|
+
# field probe (shell), and a review backend (python). They cannot import each other —
|
|
8
|
+
# different repos, and the private side's vocabulary contains identifiers that must not
|
|
9
|
+
# reach a public repo. So the shareable artifact is the CORPUS, not the classifier: each
|
|
10
|
+
# side implements in its own language and proves agreement by running these rows.
|
|
11
|
+
#
|
|
12
|
+
# Measured drift 2026-08-08, before this file existed: 19 probe ids, **10 disagreements**
|
|
13
|
+
# between two of the implementations — and asymmetrically, 9 of them were the private side
|
|
14
|
+
# reading INCAPABLE models as CAPABLE. A permissive drift on the side that faces a real
|
|
15
|
+
# model catalog is the dangerous direction: it inflates panel size and family-diversity
|
|
16
|
+
# counts simultaneously, so the defense reports green while the panel cannot review.
|
|
17
|
+
#
|
|
18
|
+
# RESIDENCY: this file is PUBLIC. Generic capability classes and publicly-documented model
|
|
19
|
+
# families only. A corp/internal model id NEVER goes here — carry those in a local overlay
|
|
20
|
+
# corpus alongside (gitignored), same two-layer shape as the public-surface pattern file.
|
|
21
|
+
#
|
|
22
|
+
# NAMESPACES — declare what you own, skip what you do not (OPA-bundle `roots` shape).
|
|
23
|
+
# Two different things get classified under the same question and they are NOT interchangeable:
|
|
24
|
+
# ns:cli sidecar / CLI names a panel is composed of — `codex`, `gemini`, `agy`
|
|
25
|
+
# ns:model-id concrete served model identifiers — `voyage-3`, `glm-ocr`
|
|
26
|
+
# A consumer that classifies model ids has no opinion on `codex`; a consumer that composes
|
|
27
|
+
# panels has no opinion on `bge-m3`. Mixing them silently is a defect: measured 2026-08-08,
|
|
28
|
+
# the first draft of this corpus put both in one list and a model-id classifier failed the
|
|
29
|
+
# `codex` row — reported as implementation drift when it was corpus scope error. An adapter
|
|
30
|
+
# MUST declare its namespaces and SKIP rows outside them EXPLICITLY (a skipped row is
|
|
31
|
+
# reported, never silently counted as passing — unmeasured is not clean).
|
|
32
|
+
#
|
|
33
|
+
# FORMAT (tab-separated): id <TAB> expected <TAB> ns:<namespace>/<class> <TAB> why
|
|
34
|
+
# expected ∈ CAPABLE | INCAPABLE | UNDECIDABLE
|
|
35
|
+
# CAPABLE a generative model of a family known to return prose findings
|
|
36
|
+
# INCAPABLE emits vectors / scores / labels / transcripts — cannot return a finding
|
|
37
|
+
# UNDECIDABLE unrecognised. NOT "capable by default" — default-deny, because a
|
|
38
|
+
# denylist cannot know that `voyage-3` is an embedding model. An
|
|
39
|
+
# implementation that answers CAPABLE here is failing open.
|
|
40
|
+
#
|
|
41
|
+
# THE CORPUS STATES GROUND TRUTH, NOT WHAT ANY IMPLEMENTATION CAN SEE. `voyage-3` IS an
|
|
42
|
+
# embedding model, so the row says INCAPABLE — even though a pattern classifier cannot
|
|
43
|
+
# tell from the string. Grading is therefore on OUTCOME SAFETY, not label identity:
|
|
44
|
+
#
|
|
45
|
+
# hard FAIL want ∈ {INCAPABLE, UNDECIDABLE} and got = CAPABLE (permissive — ships a
|
|
46
|
+
# broken panel; the only direction that can actually hurt)
|
|
47
|
+
# hard FAIL want = CAPABLE and got ≠ CAPABLE (over-block — an
|
|
48
|
+
# always-red gate teaches --no-verify)
|
|
49
|
+
# DEVIATION INCAPABLE ↔ UNDECIDABLE in either direction (reported, not failed:
|
|
50
|
+
# both BLOCK under default-deny, so the outcome is identical)
|
|
51
|
+
#
|
|
52
|
+
# This rule exists because the first draft graded on label identity and was silently fitted
|
|
53
|
+
# to one implementation's regex — `armorm` was written UNDECIDABLE and `voyage-3` INCAPABLE
|
|
54
|
+
# purely because that regex caught one and not the other. Two ids with the same property,
|
|
55
|
+
# two different standards, and the corpus was grading itself. Worse, it made "delete the
|
|
56
|
+
# row" the rational move for whoever maintained the weaker classifier.
|
|
57
|
+
#
|
|
58
|
+
# ORDERING INVARIANT the rows enforce: ineligibility must be tested BEFORE eligibility.
|
|
59
|
+
# Rows marked `overlap` match a valid family AND an incapable class simultaneously; an
|
|
60
|
+
# eligibility-first implementation admits them silently. They are the anchor for the ORDER,
|
|
61
|
+
# not for the list.
|
|
62
|
+
|
|
63
|
+
# ── CONTRACT PIN ───────────────────────────────────────────────────────────────
|
|
64
|
+
#CORPUS-VERSION 2
|
|
65
|
+
#CORPUS-SHA256 e0d8d83b9ae042e4
|
|
66
|
+
# Hash covers the DATA rows only (comments/format may be reworded without a version bump).
|
|
67
|
+
# Recompute EXACTLY this way — two ways of computing it is itself a drift source, and the
|
|
68
|
+
# first draft of this pin failed on a trailing-newline difference between two implementations:
|
|
69
|
+
# grep -E '^[^#[:space:]]' reviewer_capability_corpus.tsv | shasum -a 256 | cut -c1-16
|
|
70
|
+
# Every consumer recomputes it and compares. A mismatch is HARNESS_ERROR — not PASS, not FAIL:
|
|
71
|
+
# it means the consumer is reading a DIFFERENT corpus than the one this contract names, and a
|
|
72
|
+
# drifted local copy is the failure mode that makes every downstream instrument report healthy
|
|
73
|
+
# while nothing is actually shared. (Independent precedent: OPA bundle `roots`, test262/CNCF
|
|
74
|
+
# conformance-suite-as-contract; failure taxonomy: local-first resolution · silent skipping ·
|
|
75
|
+
# drifted duplicates.)
|
|
76
|
+
|
|
77
|
+
# ── capable: generative families ────────────────────────────────────────────────
|
|
78
|
+
codex CAPABLE ns:cli/generative GPT-family agent CLI
|
|
79
|
+
gpt-5.5 CAPABLE ns:model-id/generative GPT family
|
|
80
|
+
gemini CAPABLE ns:cli/generative Gemini family
|
|
81
|
+
qwen3-32b CAPABLE ns:model-id/generative Qwen family, generative variant
|
|
82
|
+
glm-4.6 CAPABLE ns:model-id/generative GLM family, generative variant
|
|
83
|
+
deepseek-v3 CAPABLE ns:model-id/generative DeepSeek family
|
|
84
|
+
mistral-large CAPABLE ns:model-id/generative Mistral family
|
|
85
|
+
llama-4-70b CAPABLE ns:model-id/generative Llama family
|
|
86
|
+
|
|
87
|
+
# ── incapable: named by class token ─────────────────────────────────────────────
|
|
88
|
+
text-embedding-3-large INCAPABLE ns:model-id/embedding emits vectors
|
|
89
|
+
some-reranker-v1 INCAPABLE ns:model-id/reranker emits scores
|
|
90
|
+
generic-ocr-2b INCAPABLE ns:model-id/ocr emits transcribed text, not judgment
|
|
91
|
+
acme-safeguard-8b INCAPABLE ns:model-id/safeguard classifier — emits labels
|
|
92
|
+
text-moderation-latest INCAPABLE ns:model-id/moderation classifier — emits labels
|
|
93
|
+
whisper-large-v3 INCAPABLE ns:model-id/speech speech-to-text
|
|
94
|
+
|
|
95
|
+
# ── incapable WITHOUT a class token in the name (the denylist's blind spot) ─────
|
|
96
|
+
# These are why a denylist alone is insufficient and default-deny is required: nothing in
|
|
97
|
+
# the string says "embedding". Every one is a real, publicly-documented model.
|
|
98
|
+
voyage-3 INCAPABLE ns:model-id/embedding embedding model; name carries no class token
|
|
99
|
+
bge-m3 INCAPABLE ns:model-id/embedding embedding model; name carries no class token
|
|
100
|
+
gte-large INCAPABLE ns:model-id/embedding embedding model; name carries no class token
|
|
101
|
+
all-minilm-l6-v2 INCAPABLE ns:model-id/embedding embedding model; name carries no class token
|
|
102
|
+
armorm INCAPABLE ns:model-id/reward reward model — emits a scalar; opaque name, no class token
|
|
103
|
+
starling-rm INCAPABLE ns:model-id/reward reward model — emits a scalar
|
|
104
|
+
cohere-rank-v3 INCAPABLE ns:model-id/reranker reranker named `rank`, not `rerank`
|
|
105
|
+
|
|
106
|
+
# ── overlap: matches a VALID family AND an incapable class (ordering anchor) ────
|
|
107
|
+
glm-ocr INCAPABLE ns:model-id/overlap matches `glm` (capable) and `ocr` (incapable) — order decides
|
|
108
|
+
qwen3-embedding-8b INCAPABLE ns:model-id/overlap matches `qwen` (capable) and `embed` (incapable)
|
|
109
|
+
snowflake-arctic-embed-l-v2.0 INCAPABLE ns:model-id/overlap family prefix unrecognised; `embed` decides
|
|
110
|
+
|
|
111
|
+
# ── undecidable: unrecognised → default-deny, never assumed capable ─────────────
|
|
112
|
+
some-new-model-x UNDECIDABLE ns:model-id/unknown no known family; answering CAPABLE here is fail-open
|
|
113
|
+
internal-model-7 UNDECIDABLE ns:model-id/unknown no known family
|
|
114
|
+
codeguardian UNDECIDABLE ns:cli/unknown NOT auto-incapable — a bare `guard` substring rule blocks this legitimately-named review tool; unrecognised, so default-deny applies, but as UNDECIDABLE not INCAPABLE
|
|
115
|
+
|
|
116
|
+
# ── same-family exclusion (decorrelation, not capability) ──────────────────────
|
|
117
|
+
# Capable models that must still be REJECTED as panel members when the governor is of that
|
|
118
|
+
# family. Capability and decorrelation are different axes; an implementation that only
|
|
119
|
+
# checks capability passes a same-family panel. Marked separately so an adapter that does
|
|
120
|
+
# not own the decorrelation axis can skip this block explicitly rather than silently.
|
|
121
|
+
#SAMEFAMILY claude
|
|
122
|
+
#SAMEFAMILY opus
|
|
123
|
+
#SAMEFAMILY sonnet
|
|
124
|
+
#SAMEFAMILY haiku
|
package/scripts/selfcheck.sh
CHANGED
|
@@ -25,6 +25,74 @@ check() { # check <label> <cmd...>
|
|
|
25
25
|
fail=1
|
|
26
26
|
fi
|
|
27
27
|
}
|
|
28
|
+
# ⚠️ KNOWN DEFECT, NOT FIXED HERE — `check()` above has the same evidence-discarding shape the
|
|
29
|
+
# lane blocks below were repaired for (2026-08-05): it decides on `"$@" 2>/dev/null` (stderr of the
|
|
30
|
+
# DECIDING run is destroyed) and then re-runs to print. It is left alone deliberately: `check()` is
|
|
31
|
+
# called by every `node --check` / `bash -n` line in this file, so changing it changes the whole
|
|
32
|
+
# surface at once, which is a different job from repairing the four lane blocks (CLAUDE.md
|
|
33
|
+
# §Added-Scope Gate question 2). `scripts/probe_scope_check.sh`'s caller near the probe-scope block
|
|
34
|
+
# carries the same shape. Both are tracked separately — do NOT read the lane-block repair below as
|
|
35
|
+
# having cleared this file.
|
|
36
|
+
|
|
37
|
+
# _show_failure <captured-output> — print a FAILING suite's evidence without truncating it away.
|
|
38
|
+
# Single source for all four lane blocks (a second copy would drift; the divergent-normalizer class).
|
|
39
|
+
# WHY NOT `tail -N`: measured 2026-08-05 on sync_from_be_lanes.sh — output is 98 lines and a planted
|
|
40
|
+
# lane failure at line ~22 is INVISIBLE to `tail -20` (0 hits), while the summary banner still reads
|
|
41
|
+
# "1 failed". The reader gets a FAIL verdict sitting on top of passing log lines — the exact shape
|
|
42
|
+
# this whole repair exists to remove. The failing-line extraction finds it (1 hit, known pair).
|
|
43
|
+
# All four suites mark failures with `❌` (`no()` in sync_from_be_lanes.sh:21, `chk()` in the other
|
|
44
|
+
# three) or an early `FAIL ` line when a subject is missing; grep handles the multi-byte glyph
|
|
45
|
+
# (known pair: 1 hit on a ❌ line, 0 on a ✅-only line — verified, not assumed).
|
|
46
|
+
_show_failure() {
|
|
47
|
+
local out="$1" fails n banner shown nonblank
|
|
48
|
+
# Whitespace-only counts as empty: guarding with [ -z "$out" ] alone let a suite emitting " "
|
|
49
|
+
# fall into the died-early branch and print indented blank lines — silence rendered as evidence.
|
|
50
|
+
# SHELL PATTERN, NOT `tr`: the obvious `tr -d '[:space:]'` is a measured defect on BSD. Given a
|
|
51
|
+
# line containing invalid UTF-8, macOS `tr` aborts with `tr: Illegal byte sequence` and emits
|
|
52
|
+
# NOTHING, so the guard concludes "empty" and reports "no output captured" while real evidence is
|
|
53
|
+
# sitting in $out — the exact mis-report this helper exists to prevent, reintroduced by the guard
|
|
54
|
+
# against it. (GNU tr passes the bytes through; the arms disagree, and the failing arm is the
|
|
55
|
+
# author's own machine.) Case-matching is a shell builtin: no subprocess, no charset decoding, so
|
|
56
|
+
# invalid bytes cannot make it lie. Known pair: invalid-byte→nonblank, spaces/tabs/newlines→blank,
|
|
57
|
+
# ""→blank, "hello"→nonblank (4/4), while the tr form returns 0 bytes on arm 1.
|
|
58
|
+
case "$out" in *[![:space:]]*) nonblank=1 ;; *) nonblank= ;; esac
|
|
59
|
+
# NO LOCALE PIN HERE, and its absence is a measured result — same disposition, and same reasoning,
|
|
60
|
+
# as the pin `.github/workflows/validate.yml` removed after refuting its own locale hypothesis.
|
|
61
|
+
# Two model families independently suspected that matching the multi-byte `❌` would break under a
|
|
62
|
+
# C/POSIX locale (one filed it as UNCALIBRATED for the GNU arm, the other as an unpinned-locale
|
|
63
|
+
# defect), so `LC_ALL=C` was added — then both arms were actually measured:
|
|
64
|
+
# BSD grep (macOS) : C, UTF-8 → 1 hit each
|
|
65
|
+
# GNU grep 3.12 (Linux) : C, C.UTF-8, unset → 1 hit each; and with invalid UTF-8 bytes mixed
|
|
66
|
+
# in, the ❌ line still extracts (no binary-file
|
|
67
|
+
# collapse, the specific feared mode)
|
|
68
|
+
# The hypothesis is REFUTED on both arms, so the pin demonstrated nothing and was removed rather
|
|
69
|
+
# than kept as insurance — a knob retained because it might help is indistinguishable from one
|
|
70
|
+
# that does, and the next reader would inherit it as evidence that the danger is real.
|
|
71
|
+
fails=$(printf '%s\n' "$out" | grep -E '❌|^FAIL ' || true)
|
|
72
|
+
banner=$(printf '%s\n' "$out" | grep -E '════' | tail -1 || true)
|
|
73
|
+
if [ -n "$fails" ]; then
|
|
74
|
+
shown=$(printf '%s\n' "$fails" | head -25)
|
|
75
|
+
printf '%s\n' "$shown" | sed 's/^/ /'
|
|
76
|
+
n=$(printf '%s\n' "$fails" | wc -l | tr -d ' ')
|
|
77
|
+
[ "$n" -gt 25 ] && echo " … ($((n - 25)) more failing lines not shown)"
|
|
78
|
+
elif [ -z "$nonblank" ]; then
|
|
79
|
+
echo " (no output captured — the suite produced nothing before failing)"
|
|
80
|
+
else
|
|
81
|
+
echo " (no ❌/FAIL line found — suite likely died early; showing tail)"
|
|
82
|
+
printf '%s\n' "$out" | tail -12 | sed 's/^/ /'
|
|
83
|
+
fi
|
|
84
|
+
# Summary banner: matched by shape, not by position. A blind `tail -2` re-printed lines already
|
|
85
|
+
# shown above (measured: a ❌ within the last 2 lines appeared twice, and the "N more not shown"
|
|
86
|
+
# notice was immediately followed by one of the lines it had just declined to show), and on empty
|
|
87
|
+
# input it emitted a stray indented line. Print it only when it exists and is not already on screen.
|
|
88
|
+
# Dedupe against what was ACTUALLY PRINTED ($shown), not against the full $fails set. Searching
|
|
89
|
+
# $fails suppressed the banner whenever a failing line beyond the head -25 cut merely contained
|
|
90
|
+
# the banner text — i.e. it hid the banner precisely because a line the reader never saw mentioned
|
|
91
|
+
# it. $shown is empty in the non-fails branches, so the banner prints there as before.
|
|
92
|
+
if [ -n "$banner" ] && ! printf '%s\n' "${shown:-}" | grep -qF -- "$banner"; then
|
|
93
|
+
printf ' %s\n' "$banner"
|
|
94
|
+
fi
|
|
95
|
+
}
|
|
28
96
|
|
|
29
97
|
# Node executables (npm-shipped)
|
|
30
98
|
for f in bin/*.js; do
|
|
@@ -34,7 +102,18 @@ done
|
|
|
34
102
|
# Codex adapter drift: the thin Codex runtime must keep reading canonical FH
|
|
35
103
|
# skill/agent surfaces without silently accepting Claude-native primitives as
|
|
36
104
|
# Codex-native.
|
|
37
|
-
check
|
|
105
|
+
# NOT via check(): that helper decides on a run whose stderr is discarded, and this call site used to
|
|
106
|
+
# additionally discard the subject's STDOUT (`--strict >/dev/null`). fh-codex-doctor writes 100% of
|
|
107
|
+
# its drift diagnostics to stdout (measured 2026-08-05: 686 B stdout / 0 B stderr), so a failure
|
|
108
|
+
# printed a bare `FAIL` line carrying no diagnosis at all — worse than the truncation this session
|
|
109
|
+
# repaired in the lane blocks. Fixed at the call site; check() itself is a separate job (see above).
|
|
110
|
+
if _out=$(node bin/fh-codex-doctor.js --strict 2>&1); then
|
|
111
|
+
echo "PASS fh-codex-doctor --strict"
|
|
112
|
+
else
|
|
113
|
+
echo "FAIL fh-codex-doctor --strict"
|
|
114
|
+
_show_failure "$_out"
|
|
115
|
+
fail=1
|
|
116
|
+
fi
|
|
38
117
|
|
|
39
118
|
# Bash surface: npm-shipped scripts + local bin wrappers + gate-chain infra
|
|
40
119
|
for f in scripts/*.sh bin/fh-gate bin/fh-run bin/fh-goal \
|
|
@@ -132,6 +211,45 @@ else
|
|
|
132
211
|
fail=1
|
|
133
212
|
fi
|
|
134
213
|
|
|
214
|
+
# embedded --self-test suites (compaction_probe · judgment_circuit_lint · novelty_claim_check).
|
|
215
|
+
# These three carry their lanes INSIDE the script (`--self-test`) rather than in a sibling
|
|
216
|
+
# test_*_lanes.sh, so the name-list wiring above skipped them silently: 48 lanes existed and ran
|
|
217
|
+
# only when a human typed the command. That is built-but-not-wired applied to the anchors themselves
|
|
218
|
+
# — a later edit that breaks a lane stays green everywhere the project actually checks
|
|
219
|
+
# (high re-review 2026-08-08). Same shape as the block above: subject absent → SKIP, subject present
|
|
220
|
+
# but self-test missing → FAIL, never a silent pass.
|
|
221
|
+
for _subj in compaction_probe judgment_circuit_lint novelty_claim_check; do
|
|
222
|
+
if [ ! -f "scripts/$_subj.sh" ]; then
|
|
223
|
+
echo "SKIP $_subj --self-test (subject scripts/$_subj.sh absent)"
|
|
224
|
+
else
|
|
225
|
+
# ⚠️ **문자열 존재로 판정하지 마라.** 초판은 `grep -q -- '--self-test'` 였는데, 그 문자열은
|
|
226
|
+
# 헤더 주석과 usage echo 에도 있어서 **디스패처 한 줄만 지워도 여전히 매치**한다. 그리고
|
|
227
|
+
# 인식 못 한 모드에서 스크립트가 usage 를 찍고 exit 0 을 내므로, selfcheck 는 rc=0 을 보고
|
|
228
|
+
# 조용히 통과했다 — 25개 레인이 통째로 사라져도 `npm test` 는 PASS (high 3차 리뷰 실측).
|
|
229
|
+
# 존재검사가 진위를 못 본다는 그 클래스의 재발이다. **실행이 일어났다는 증거**를 요구한다.
|
|
230
|
+
# `< /dev/null` 필수: 인식 못 한 모드로 떨어지면 스크립트가 stdin 을 기다려 **무한 대기**한다
|
|
231
|
+
# (실측 — 디스패처 제거 known-negative 가 2분 타임아웃). CI 를 멈추는 건 조용한 통과보다 나쁘다.
|
|
232
|
+
# `timeout` 은 GNU coreutils 이고 **stock macOS 에 없다**. 가용성 확인 없이 부르면 rc=127 +
|
|
233
|
+
# 빈 출력 → 아래 `*)` 가 발동해 "dispatcher missing?" 이라는 **틀린 원인**으로 거짓 FAIL 이
|
|
234
|
+
# 난다(실측: homebrew 없는 PATH 에서 3개 subject 전부). 소비자 머신에서 `npm test` 와
|
|
235
|
+
# `prepublishOnly` 를 깨뜨리는 경로다. 정답 폼은 이미 레포에 있다(sync-from-be.sh:134).
|
|
236
|
+
# 없으면 무한대기 방지를 잃는 대신 도는 쪽을 택한다 — `< /dev/null` 이 그 방어의 본체다.
|
|
237
|
+
local _to=""; command -v timeout >/dev/null 2>&1 && _to="timeout 120"
|
|
238
|
+
_st_out="$($_to bash "scripts/$_subj.sh" --self-test < /dev/null 2>&1)"; _st_rc=$?
|
|
239
|
+
case "$_st_out" in
|
|
240
|
+
*캘리브레이션*) : ;;
|
|
241
|
+
*) echo "FAIL $_subj: --self-test produced no calibration verdict (dispatcher missing?)"
|
|
242
|
+
printf '%s\n' "$_st_out" | head -3 | sed 's/^/ /'
|
|
243
|
+
fail=1; _st_rc=0 ;; # 이미 FAIL 로 셌으니 아래서 중복 계상 안 한다
|
|
244
|
+
esac
|
|
245
|
+
if [ "$_st_rc" -ne 0 ]; then
|
|
246
|
+
echo "FAIL $_subj --self-test (exit $_st_rc)"
|
|
247
|
+
printf '%s\n' "$_st_out" | tail -6 | sed 's/^/ /'
|
|
248
|
+
fail=1
|
|
249
|
+
fi
|
|
250
|
+
fi
|
|
251
|
+
done
|
|
252
|
+
|
|
135
253
|
# memory-link-check — the memory store is a GRAPH (memory_intent_recall.md: nodes=files,
|
|
136
254
|
# edges=[[links]], recall walks one hop). Measured 2026-07-28: 50 of 872 edges pointed at a note
|
|
137
255
|
# that existed under a different separator and 22 at nothing — a dead edge returns nothing and is
|
|
@@ -441,34 +559,81 @@ else
|
|
|
441
559
|
fail=1
|
|
442
560
|
fi
|
|
443
561
|
|
|
562
|
+
# utterance landing check — the close chain's CONTENT anchor, and it has to be RUN, not parsed.
|
|
563
|
+
# Measured 2026-08-08 on the branch that introduced it: the only thing in this file that touched
|
|
564
|
+
# `scripts/utterance_landing_check.sh` was the `bash -n` syntax sweep at the top. Syntax passing is
|
|
565
|
+
# not the instrument working, so the script shipped via files[] with zero behavioural callers —
|
|
566
|
+
# [[feedback_built_but_not_wired]] in its purest form, where the sole caller is prose in CLAUDE.md.
|
|
567
|
+
#
|
|
568
|
+
# What the self-test defends is specifically the DEGRADE DIRECTION. This checker's whole reason for
|
|
569
|
+
# existing is that a dead grep prints zero hits and zero hits read as "nothing landed" — a fail-open
|
|
570
|
+
# that manufactures a clean verdict out of a broken instrument. Its known-pair set pins that apart:
|
|
571
|
+
# a genuine miss must exit 1 while a dead control must exit 10, and 10 must never collapse into 1.
|
|
572
|
+
# Left unrun, the file rots exactly where it is load-bearing and nothing here would notice.
|
|
573
|
+
# Absence is a FAIL, not a SKIP — and the distinction is mechanical, not stylistic. The SKIP arm
|
|
574
|
+
# above for `probe_scope_check.sh` is correct because that file genuinely never ships; this one is
|
|
575
|
+
# listed in package.json files[], so in BOTH a source tree and an installed package it must be here.
|
|
576
|
+
# A SKIP would convert `rm scripts/utterance_landing_check.sh` into a green run — deletion as a
|
|
577
|
+
# clean bill of health, which is the same fail-open shape the checker itself exists to refuse.
|
|
578
|
+
if [ ! -f scripts/utterance_landing_check.sh ]; then
|
|
579
|
+
echo "FAIL utterance_landing_check.sh: missing — it ships via package.json files[], so absence is deletion, not package mode"
|
|
580
|
+
fail=1
|
|
581
|
+
elif _out=$(bash scripts/utterance_landing_check.sh --self-test 2>&1); then
|
|
582
|
+
echo "PASS utterance_landing_check.sh (known-pair self-test: control-death → 10, target-miss → 1)"
|
|
583
|
+
else
|
|
584
|
+
echo "FAIL utterance_landing_check.sh: self-test failed — the close-chain content anchor cannot be trusted"
|
|
585
|
+
_show_failure "$_out"
|
|
586
|
+
fail=1
|
|
587
|
+
fi
|
|
588
|
+
|
|
444
589
|
# tag/version consistency guard. Wired with the guard it anchors: the guard exists because a wrong
|
|
445
590
|
# tag reached the remote and a publish from the wrong tree was stopped only by npm's own collision
|
|
446
591
|
# check, so an unrun anchor here would be the same luck-as-floor arrangement one layer up.
|
|
447
592
|
if [ ! -f templates/.git-hooks/pre-push ]; then
|
|
448
593
|
echo "SKIP test_tag_version_lanes.sh (subject templates/.git-hooks/pre-push absent)"
|
|
449
594
|
elif [ -f scripts/test_tag_version_lanes.sh ]; then
|
|
450
|
-
|
|
595
|
+
# RUN-ONCE, CAPTURE (2026-08-05) — rationale in the sync_from_be_lanes block later in this file.
|
|
596
|
+
if _out=$(bash scripts/test_tag_version_lanes.sh 2>&1); then
|
|
597
|
+
echo "PASS test_tag_version_lanes.sh (mismatch blocks · match silent · scope · override)"
|
|
598
|
+
else
|
|
451
599
|
echo "FAIL test_tag_version_lanes.sh: the tag/version guard would mis-route"
|
|
452
|
-
|
|
600
|
+
_show_failure "$_out"
|
|
453
601
|
fail=1
|
|
454
|
-
else
|
|
455
|
-
echo "PASS test_tag_version_lanes.sh (mismatch blocks · match silent · scope · override)"
|
|
456
602
|
fi
|
|
457
603
|
else
|
|
458
604
|
echo "FAIL test_tag_version_lanes.sh: pre-push present but its anchor is missing"
|
|
459
605
|
fail=1
|
|
460
606
|
fi
|
|
461
607
|
|
|
608
|
+
# Shipped-manifest version lockstep. Distinct from the tag lane above: that one compares the git TAG
|
|
609
|
+
# to package.json; this one compares package.json to every version string it SHIPS — including the
|
|
610
|
+
# per-plugin entries inside marketplace.json, which the tag lane never opens. Measured 2026-08-06:
|
|
611
|
+
# a bump left the second marketplace entry behind and the tag lane passed 8/8 straight through it.
|
|
612
|
+
if [ ! -f scripts/version_lockstep_check.sh ]; then
|
|
613
|
+
echo "SKIP test_version_lockstep_lanes.sh (subject scripts/version_lockstep_check.sh absent)"
|
|
614
|
+
elif [ -f scripts/test_version_lockstep_lanes.sh ]; then
|
|
615
|
+
if _out=$(bash scripts/test_version_lockstep_lanes.sh 2>&1); then
|
|
616
|
+
echo "PASS test_version_lockstep_lanes.sh (drift blocks · aligned silent · unreadable = exit 2, not pass)"
|
|
617
|
+
else
|
|
618
|
+
echo "FAIL test_version_lockstep_lanes.sh: the shipped-manifest lockstep guard would mis-route"
|
|
619
|
+
_show_failure "$_out"
|
|
620
|
+
fail=1
|
|
621
|
+
fi
|
|
622
|
+
else
|
|
623
|
+
echo "FAIL test_version_lockstep_lanes.sh: version_lockstep_check.sh present but its anchor is missing"
|
|
624
|
+
fail=1
|
|
625
|
+
fi
|
|
626
|
+
|
|
462
627
|
# ④-e dispatch-log reconciliation + its tally hook. Wired in the same commit that ships them: the
|
|
463
628
|
# obligation they mechanize lost 20/20 in a single session, so leaving the checker itself unrun
|
|
464
629
|
# would be the same defect one layer up.
|
|
465
630
|
if [ -f scripts/test_dispatch_log_lanes.sh ]; then
|
|
466
|
-
if
|
|
631
|
+
if _out=$(bash scripts/test_dispatch_log_lanes.sh 2>&1); then
|
|
632
|
+
echo "PASS test_dispatch_log_lanes.sh (date-spelling + verdict + tally-hook lanes)"
|
|
633
|
+
else
|
|
467
634
|
echo "FAIL test_dispatch_log_lanes.sh: the dispatch-log reconciliation would mis-report"
|
|
468
|
-
|
|
635
|
+
_show_failure "$_out"
|
|
469
636
|
fail=1
|
|
470
|
-
else
|
|
471
|
-
echo "PASS test_dispatch_log_lanes.sh (date-spelling + verdict + tally-hook lanes)"
|
|
472
637
|
fi
|
|
473
638
|
fi
|
|
474
639
|
|
|
@@ -477,12 +642,12 @@ fi
|
|
|
477
642
|
# through in silence, then a package discriminator keyed on a file that actually ships. Both are
|
|
478
643
|
# known-POSITIVEs in the suite, so neither can come back green.
|
|
479
644
|
if [ -f scripts/test_selfcheck_state_lanes.sh ]; then
|
|
480
|
-
if
|
|
645
|
+
if _out=$(bash scripts/test_selfcheck_state_lanes.sh 2>&1); then
|
|
646
|
+
echo "PASS test_selfcheck_state_lanes.sh (four input states + both shipped mis-routings)"
|
|
647
|
+
else
|
|
481
648
|
echo "FAIL test_selfcheck_state_lanes.sh: a subject-presence discriminator would mis-route"
|
|
482
|
-
|
|
649
|
+
_show_failure "$_out"
|
|
483
650
|
fail=1
|
|
484
|
-
else
|
|
485
|
-
echo "PASS test_selfcheck_state_lanes.sh (four input states + both shipped mis-routings)"
|
|
486
651
|
fi
|
|
487
652
|
fi
|
|
488
653
|
|
|
@@ -495,12 +660,34 @@ fi
|
|
|
495
660
|
if [ ! -f scripts/sync-from-be.sh ]; then
|
|
496
661
|
echo "SKIP sync_from_be_lanes.sh (subject scripts/sync-from-be.sh absent)"
|
|
497
662
|
elif [ -f scripts/sync_from_be_lanes.sh ]; then
|
|
498
|
-
|
|
663
|
+
# ── RUN-ONCE, CAPTURE — canonical note for the four LANE BLOCKS (2026-08-05) ─────────────────
|
|
664
|
+
# SCOPE, stated precisely because the first draft of this note over-claimed: this covers the four
|
|
665
|
+
# lane blocks only (tag-version · dispatch-log · selfcheck-state · sync_from_be). The same
|
|
666
|
+
# evidence-discarding shape SURVIVES in `check()` at the top of this file and in the
|
|
667
|
+
# probe_scope_check caller — both named there, both deliberately out of scope, both still open.
|
|
668
|
+
# An adversarial round caught the original "all four sites in this file" wording as a false
|
|
669
|
+
# completion claim: it would have stopped the next reader from re-searching. Half a fix with a
|
|
670
|
+
# done-label on it is worse than half a fix.
|
|
671
|
+
# The old form ran the suite twice: once discarded to /dev/null to decide, once re-run to print.
|
|
672
|
+
# For a DETERMINISTIC suite that is merely wasteful. For a non-deterministic one it destroys the
|
|
673
|
+
# evidence: the failing run's output goes to /dev/null and the reader is shown the SECOND run,
|
|
674
|
+
# which may pass. Measured here 2026-08-04 (run 30955950695) — CI printed
|
|
675
|
+
# FAIL sync_from_be_lanes.sh: return-path lanes failed
|
|
676
|
+
# ════ lanes: 70 passed · 0 failed ════
|
|
677
|
+
# i.e. a FAIL verdict over a PASSING transcript, and the actual failure was never recorded
|
|
678
|
+
# anywhere. That is why this lane sat "flaky, cause unknown" on the session card for two days:
|
|
679
|
+
# the instrument was discarding the only evidence that could close it. Reproduced as a known
|
|
680
|
+
# pair before this edit (arm A run-twice → evidence lost + self-contradiction; arm B run-once →
|
|
681
|
+
# evidence preserved), so the fix is anchored, not asserted.
|
|
682
|
+
# NOTE this does NOT make the suite deterministic — the underlying non-determinism is still
|
|
683
|
+
# UNDIAGNOSED and stays an open item. It makes the next occurrence diagnosable instead of
|
|
684
|
+
# self-erasing. Do not read a green CI after this change as the flake being fixed.
|
|
685
|
+
if _out=$(bash scripts/sync_from_be_lanes.sh 2>&1); then
|
|
686
|
+
echo "PASS sync_from_be_lanes.sh (return-path lanes)"
|
|
687
|
+
else
|
|
499
688
|
echo "FAIL sync_from_be_lanes.sh: return-path lanes failed"
|
|
500
|
-
|
|
689
|
+
_show_failure "$_out"
|
|
501
690
|
fail=1
|
|
502
|
-
else
|
|
503
|
-
echo "PASS sync_from_be_lanes.sh (return-path lanes)"
|
|
504
691
|
fi
|
|
505
692
|
else
|
|
506
693
|
echo "FAIL sync_from_be_lanes.sh: sync-from-be.sh present but its anchor is missing"
|
|
@@ -15,6 +15,9 @@
|
|
|
15
15
|
|
|
16
16
|
set -uo pipefail
|
|
17
17
|
|
|
18
|
+
# 상속된 git 환경변수를 끊는다 — export 된 GIT_DIR 이 있으면 `git -C "$FH"` 가 인자로 받은
|
|
19
|
+
# 레포가 아니라 그 레포를 잰다(Axis 2 at-floor LOW, 2026-08-06). READ-ONLY 체커라 부작용 없음.
|
|
20
|
+
unset GIT_DIR GIT_WORK_TREE GIT_INDEX_FILE
|
|
18
21
|
FH="${1:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"
|
|
19
22
|
TODAY=$(date +%Y-%m-%d)
|
|
20
23
|
CARD="$FH/tracks/_meta/reference_next_session_starter.md"
|
|
@@ -25,11 +28,58 @@ _mtime() { stat -c %Y "$1" 2>/dev/null || stat -f %m "$1" 2>/dev/null || echo 0;
|
|
|
25
28
|
echo "── session close check: $FH ($TODAY) ──"
|
|
26
29
|
|
|
27
30
|
# ① status snapshot — uncommitted / unpushed work must be known, not forgotten
|
|
28
|
-
DIRTY
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
31
|
+
# DIRTY 도 같은 형태였다 — `status --porcelain 2>/dev/null | wc -l` 은 git 이 죽어도 0 을 세어
|
|
32
|
+
# "깨끗함"으로 승격된다(재현: `.git/index` 손상 → exit 128, 출력 0줄). 종료코드를 먼저 본다.
|
|
33
|
+
# ⚠️ 이 줄은 **바로 아래 UNPUSHED 수리와 같은 결함**이었고, 처음엔 아래만 고쳤다 —
|
|
34
|
+
# 반쪽 수리를 고치는 커밋에서 반쪽 수리를 할 뻔했다(Axis 2 챌린저가 잡음, 2026-08-06).
|
|
35
|
+
# `--untracked-files=all` 은 `status.showUntrackedFiles=no` config 를 덮어쓴다. 그 config 아래에서는
|
|
36
|
+
# git 이 **성공적으로 침묵**해(exit 0 · 빈 출력) 종료코드 가드로도 안 잡힌다 — 부재가 다시
|
|
37
|
+
# "깨끗함"으로 렌더된다(Axis 2 at-floor MED, 재현: 미추적 파일 1건이 0 으로 보고됨).
|
|
38
|
+
if _st=$(git -C "$FH" status --porcelain --untracked-files=all 2>/dev/null); then
|
|
39
|
+
DIRTY_KNOWN=1
|
|
40
|
+
DIRTY=$(printf '%s' "$_st" | grep -c . || true)
|
|
41
|
+
else
|
|
42
|
+
DIRTY_KNOWN=0
|
|
43
|
+
DIRTY=0
|
|
44
|
+
fi
|
|
45
|
+
# upstream 유무를 **먼저 판정**한다. `@{u}..` 는 upstream 이 없으면 실패해 0줄을 내고, 그 0을
|
|
46
|
+
# `wc -l` 이 0으로 세어 "nothing unpushed" 로 승격된다 — **한 번도 머신을 떠난 적 없는 커밋이
|
|
47
|
+
# '푸시할 것 없음'으로 읽히는 fail-open**. 부재를 깨끗함으로 읽는 것이라 0 을 신뢰하면 안 된다.
|
|
48
|
+
# (downstream fork's review lane caught it first; this file is the upstream original — 2026-08-06.)
|
|
49
|
+
if git -C "$FH" rev-parse --abbrev-ref '@{u}' >/dev/null 2>&1; then
|
|
50
|
+
UPSTREAM_KNOWN=1
|
|
51
|
+
UNPUSHED=$(git -C "$FH" log --oneline @{u}.. 2>/dev/null | wc -l | tr -d ' ')
|
|
52
|
+
else
|
|
53
|
+
UPSTREAM_KNOWN=0
|
|
54
|
+
UNPUSHED=0
|
|
55
|
+
fi
|
|
56
|
+
[ "$DIRTY_KNOWN" -eq 0 ] && echo "⚠️ ① UNMEASURED — git status failed; working-tree cleanliness is UNKNOWN, not clean"
|
|
57
|
+
[ "$DIRTY_KNOWN" -eq 1 ] && [ "$DIRTY" -gt 0 ] && echo "⚠️ ① $DIRTY uncommitted path(s) — decide: commit or leave deliberately"
|
|
58
|
+
[ "$UPSTREAM_KNOWN" -eq 0 ] && echo "⚠️ ① UNMEASURED — no upstream for this branch; unpushed count is UNKNOWN, not zero"
|
|
59
|
+
[ "$UPSTREAM_KNOWN" -eq 1 ] && [ "$UNPUSHED" -gt 0 ] && echo "⚠️ ① $UNPUSHED unpushed commit(s) — push before close or record why"
|
|
60
|
+
# "as of last fetch" — 원격을 조회하지 않는다. 로컬 remote-tracking ref 가 낡았으면 이 0 도 낡은 값이다
|
|
61
|
+
# (Axis 2 챌린저 MED, 2026-08-06). fetch 를 넣지 않은 것은 마감 체커가 READ-ONLY·오프라인 안전이기 때문.
|
|
62
|
+
# 추적 파일이 assume-unchanged/skip-worktree 로 마킹돼 있으면 그 수정은 porcelain 에 **안 뜬다** —
|
|
63
|
+
# `-uall` 로도 안 잡히는 별개 계기다(Axis 2 at-floor MED, 재현 확인).
|
|
64
|
+
MASKED=$(git -C "$FH" ls-files -v 2>/dev/null | grep -c '^[a-z]' || true)
|
|
65
|
+
[ "${MASKED:-0}" -gt 0 ] \
|
|
66
|
+
&& echo "⚠️ ① $MASKED file(s) assume-unchanged/skip-worktree — their edits are INVISIBLE here"
|
|
67
|
+
|
|
68
|
+
# ★ 잰 범위 ≠ 주장 범위 (Axis 2 at-floor HIGH, 2026-08-06).
|
|
69
|
+
# `@{u}..` 는 **현재 브랜치만** 잰다. 다른 로컬 브랜치에만 있는 미푸시 커밋은 통째로 안 보이는데
|
|
70
|
+
# 화면 문구는 "nothing unpushed"(레포 전체)라고 말한다 — 이 파일의 존재 이유 정중앙이다.
|
|
71
|
+
# 재현: 다른 브랜치에 미푸시 커밋 1건 → `✅ nothing unpushed` 가 그대로 떴다.
|
|
72
|
+
# 원격이 하나도 없으면 이 값이 전 커밋 수로 부풀므로 원격 존재를 먼저 가드한다.
|
|
73
|
+
OTHER_UNPUSHED=0
|
|
74
|
+
if [ -n "$(git -C "$FH" remote 2>/dev/null)" ]; then
|
|
75
|
+
OTHER_UNPUSHED=$(git -C "$FH" log --branches --not --remotes --oneline 2>/dev/null | wc -l | tr -d ' ')
|
|
76
|
+
fi
|
|
77
|
+
[ "${OTHER_UNPUSHED:-0}" -gt 0 ] \
|
|
78
|
+
&& echo "⚠️ ① $OTHER_UNPUSHED commit(s) on local branches never pushed anywhere (all-branch scan)"
|
|
79
|
+
|
|
80
|
+
[ "$DIRTY_KNOWN" -eq 1 ] && [ "$DIRTY" -eq 0 ] && [ "$UPSTREAM_KNOWN" -eq 1 ] && [ "$UNPUSHED" -eq 0 ] \
|
|
81
|
+
&& [ "${OTHER_UNPUSHED:-0}" -eq 0 ] && [ "${MASKED:-0}" -eq 0 ] \
|
|
82
|
+
&& echo "✅ ① working tree clean, nothing unpushed anywhere (as of last fetch)"
|
|
33
83
|
|
|
34
84
|
# ①-b open-PR sweep (surface-not-auto — requires gh; skip silently offline)
|
|
35
85
|
if command -v gh >/dev/null 2>&1; then
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# test_marker_crossfamily_lanes.sh — regression fixtures for pre-commit's
|
|
3
|
+
# validate_crossfamily_leg (typed cross-family verdict, 2026-08-08).
|
|
4
|
+
#
|
|
5
|
+
# WHY: the `crossfamily:` marker line has been REQUIRED on load-bearing changes since
|
|
6
|
+
# 2026-06 (commit c1fa459) — but only its PRESENCE was checked, so the value was free
|
|
7
|
+
# prose. That is how a false one shipped: a sibling harness recorded
|
|
8
|
+
# `crossfamily: none this round — 도달 불가` (unreachable), the claim was later found
|
|
9
|
+
# FALSE, and a later session cited it as grounds. Presence-checking catches silence; it
|
|
10
|
+
# cannot catch a confident wrong answer. This lane types the value.
|
|
11
|
+
#
|
|
12
|
+
# Fixtures assert BOTH directions (known-pair): every intended shape is admitted, and
|
|
13
|
+
# every hole the lane closes still blocks. A test that only asserts BLOCK cannot tell
|
|
14
|
+
# "blocked" from "blocked for the wrong reason"; one that only asserts PASS cannot see a
|
|
15
|
+
# lane that admits everything.
|
|
16
|
+
#
|
|
17
|
+
# Usage: bash scripts/test_marker_crossfamily_lanes.sh Exit: 0 = all behave; 1 = regression.
|
|
18
|
+
|
|
19
|
+
set -uo pipefail
|
|
20
|
+
REPO_ROOT="$(git rev-parse --show-toplevel 2>/dev/null || pwd)"
|
|
21
|
+
HOOK="$REPO_ROOT/templates/.git-hooks/pre-commit"
|
|
22
|
+
T=$(mktemp -d); trap 'rm -rf "$T"' EXIT
|
|
23
|
+
|
|
24
|
+
sed -n '/^validate_crossfamily_leg()/,/^}/p' "$HOOK" > "$T/fn.sh"
|
|
25
|
+
|
|
26
|
+
# Instrument calibration: an empty extraction would let every fixture "pass" against
|
|
27
|
+
# nothing. Assert the function body actually arrived before measuring anything.
|
|
28
|
+
if ! grep -q 'DEGRADED_PANEL_UNUSED' "$T/fn.sh"; then
|
|
29
|
+
echo "❌ HARNESS-ERROR — validate_crossfamily_leg did not extract from $HOOK."
|
|
30
|
+
echo " Fixtures below would measure an empty function. Aborting rather than reporting green."
|
|
31
|
+
exit 1
|
|
32
|
+
fi
|
|
33
|
+
|
|
34
|
+
mk() { printf "$1" > "$T/$2"; } # $1 = marker body, $2 = fixture name
|
|
35
|
+
run() { bash -c "source '$T/fn.sh'; validate_crossfamily_leg '$1'" >/dev/null 2>&1; }
|
|
36
|
+
|
|
37
|
+
FAIL=0; N=0
|
|
38
|
+
check() { # $1=fixture $2=expected(PASS|BLOCK) $3=label
|
|
39
|
+
N=$((N+1))
|
|
40
|
+
if run "$T/$1"; then got=PASS; else got=BLOCK; fi
|
|
41
|
+
if [ "$got" = "$2" ]; then echo "✅ $3 → $got"; else echo "❌ $3 → $got (expected $2)"; FAIL=1; fi
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
echo "── admitted shapes (must PASS) ──"
|
|
45
|
+
mk 'crossfamily: panel(codex,gemini) — R1..R3, 9 findings, 8 fixed 1 refuted, CONVERGED\n' k1
|
|
46
|
+
check k1 PASS "panel() with family list + verdict"
|
|
47
|
+
mk 'crossfamily: panel(codex, gemini) — spaced list, the S-grade over-block\n' k1b
|
|
48
|
+
check k1b PASS "panel(codex, gemini) — SPACE after comma (was truncated to 'panel(codex,')"
|
|
49
|
+
mk 'crossfamily: declined\n' k3
|
|
50
|
+
check k3 PASS "declined (operator chose this floor — not a degrade)"
|
|
51
|
+
mk 'crossfamily: DEGRADED_SINGLE_FAMILY — probed codex/agy/4090, 0 capable reachable\n' k4
|
|
52
|
+
check k4 PASS "DEGRADED_SINGLE_FAMILY + substantive reason"
|
|
53
|
+
mk 'crossfamily: UNKNOWN — consent unset, panel not probed this round\n' k5
|
|
54
|
+
check k5 PASS "UNKNOWN + reason naming why it was not probed"
|
|
55
|
+
mk 'crossfamily: DEGRADED_PANEL_UNUSED — codex/agy/gemini reachable, not recruited (scope)\n' k6
|
|
56
|
+
check k6 PASS "DEGRADED_PANEL_UNUSED + reason naming the reachable families"
|
|
57
|
+
mk 'crossfamily: panel(qwen,gpt-oss,glm) — clean, 0 findings\n' k7
|
|
58
|
+
check k7 PASS "capable families only — guard does not over-block a real panel"
|
|
59
|
+
|
|
60
|
+
echo "── closed holes (must BLOCK) ──"
|
|
61
|
+
mk 'axis2-evidence: PASS no-S\n' b1
|
|
62
|
+
check b1 BLOCK "no crossfamily line at all (silence — the original guard, still intact)"
|
|
63
|
+
mk 'crossfamily: none this round — 인라인 same-family\n' b2
|
|
64
|
+
check b2 BLOCK "'none' free prose (the shape found on disk)"
|
|
65
|
+
mk 'crossfamily: none this round — 도달 불가\n' b3
|
|
66
|
+
check b3 BLOCK "'none' asserting unreachable (the FALSE one that shipped)"
|
|
67
|
+
mk 'crossfamily: none\n' b4
|
|
68
|
+
check b4 BLOCK "bare 'none' (merges could-not / did-not / did-not-look)"
|
|
69
|
+
mk 'crossfamily: DEGRADED_SINGLE_FAMILY\n' b5
|
|
70
|
+
check b5 BLOCK "degrade value with no reason (silent degrade — the whole point)"
|
|
71
|
+
mk 'crossfamily: UNKNOWN\n' b6
|
|
72
|
+
check b6 BLOCK "UNKNOWN with no reason (unprobed rendered as zero)"
|
|
73
|
+
mk 'crossfamily: DEGRADED_PANEL_UNUSED\n' b7
|
|
74
|
+
check b7 BLOCK "PANEL_UNUSED with no reason ('did not' passing as 'could not')"
|
|
75
|
+
mk 'crossfamily: DEGRADED_SINGLE_FAMILY — degraded\n' b8
|
|
76
|
+
check b8 BLOCK "reason too vacuous (names no probe and no grounds)"
|
|
77
|
+
mk 'crossfamily: panel\n' b9
|
|
78
|
+
check b9 BLOCK "bare panel claim naming no family"
|
|
79
|
+
mk 'crossfamily: panel()\n' b10
|
|
80
|
+
check b10 BLOCK "panel() with empty family list"
|
|
81
|
+
mk 'crossfamily: DEGRADED\n' b11
|
|
82
|
+
check b11 BLOCK "near-miss token (not an enum value)"
|
|
83
|
+
mk 'crossfamily: codex/gpt-5.5 — R1..R4, 16 findings, CONVERGED\n' b12
|
|
84
|
+
check b12 BLOCK "legacy free-form engine string (pre-typed convention) now rejected"
|
|
85
|
+
|
|
86
|
+
echo "── cross-family review findings (agy/Gemini 3.1 Pro, 2026-08-08) ──"
|
|
87
|
+
mk 'crossfamily: panel(claude) — reviewed by another claude session\n' x1
|
|
88
|
+
check x1 BLOCK "panel names the AUTHOR'S own family (decorrelation gate not seeing decorrelation)"
|
|
89
|
+
mk 'crossfamily: panel(opus,sonnet) — two claude tiers\n' x2
|
|
90
|
+
check x2 BLOCK "two same-family tiers dressed as a 2-family panel"
|
|
91
|
+
mk 'crossfamily: panel(none)\n' x3
|
|
92
|
+
check x3 BLOCK "panel(none) — a non-run laundered through the parens"
|
|
93
|
+
mk 'crossfamily: panel(unprobed)\n' x4
|
|
94
|
+
check x4 BLOCK "panel(unprobed) — same laundering, other token"
|
|
95
|
+
mk 'crossfamily: single-family\n' x5
|
|
96
|
+
check x5 BLOCK "single-family on a LOAD-BEARING change (free no-ack bypass of the lane)"
|
|
97
|
+
mk 'crossfamily: UNKNOWN — 0\n' x6
|
|
98
|
+
check x6 BLOCK "grounds '0' (porous non-vacuity: bare digit passed)"
|
|
99
|
+
mk 'crossfamily: UNKNOWN — problem\n' x7
|
|
100
|
+
check x7 BLOCK "grounds 'problem' (matched on substring 'prob')"
|
|
101
|
+
mk 'crossfamily: UNKNOWN — client error\n' x8
|
|
102
|
+
check x8 BLOCK "grounds 'client error' (matched on substring 'cli')"
|
|
103
|
+
mk 'crossfamily: panel(voyage-3,bge-m3) — 2 families\n' x9
|
|
104
|
+
check x9 BLOCK "embedding models WITHOUT 'embed' in the name (denylist false negative)"
|
|
105
|
+
mk 'crossfamily: panel(armorm,starling-rm) — 2 families\n' x10
|
|
106
|
+
check x10 BLOCK "reward models emitting scalars, not findings"
|
|
107
|
+
mk 'crossfamily: panel(cohere-rank-v3) — reranker named 'rank' not 'rerank'\n' x11
|
|
108
|
+
check x11 BLOCK "reranker named 'rank' (denylist false negative)"
|
|
109
|
+
mk 'crossfamily: DEGRADED_SINGLE_FAMILY — agy sidecar daemon crashed on startup, none reachable\n' x12
|
|
110
|
+
check x12 PASS "grounds naming agy (was rejected — keyword list omitted it)"
|
|
111
|
+
|
|
112
|
+
mk 'crossfamily: single-family\ncrossfamily: DEGRADED_SINGLE_FAMILY — probed codex/agy, 0 reachable\n' x13
|
|
113
|
+
check x13 BLOCK "two crossfamily lines — appended correction shadowed by stale first (codex net-new)"
|
|
114
|
+
|
|
115
|
+
echo "── review-capability guard (pmh-dev #41 field measurement) ──"
|
|
116
|
+
mk 'crossfamily: panel(qwen,embed,embed) — 3 families\n' c1
|
|
117
|
+
check c1 BLOCK "embeddings counted as panel members (the measured false panel)"
|
|
118
|
+
mk 'crossfamily: panel(embed,rerank,ocr) — 3 families\n' c2
|
|
119
|
+
check c2 BLOCK "every member incapable — '3 families', 0 reviewers"
|
|
120
|
+
mk 'crossfamily: panel(gpt-oss,safeguard) — 2 families\n' c3
|
|
121
|
+
check c3 BLOCK "safeguard classifier padding one real family"
|
|
122
|
+
# Ordering invariant: ineligibility is tested FIRST. Both tokens ALSO match a valid
|
|
123
|
+
# family (glm-ocr → glm, qwen-embedding → qwen), so an eligibility-first implementation
|
|
124
|
+
# admits them silently. This pair anchors the ORDER, not the list.
|
|
125
|
+
mk 'crossfamily: panel(glm-ocr) — glm family\n' c4
|
|
126
|
+
check c4 BLOCK "glm-ocr — matches a valid family AND an incapable class"
|
|
127
|
+
mk 'crossfamily: panel(qwen-embedding-8b) — qwen family\n' c5
|
|
128
|
+
check c5 BLOCK "qwen-embedding — same overlap, other direction"
|
|
129
|
+
|
|
130
|
+
echo
|
|
131
|
+
if [ "$FAIL" -eq 0 ]; then echo "✅ all $N fixtures behave"; else echo "❌ regression ($N fixtures run)"; fi
|
|
132
|
+
exit "$FAIL"
|