@wooojin/forgen 0.4.13 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/CHANGELOG.md +94 -0
  3. package/README.ja.md +4 -4
  4. package/README.md +89 -27
  5. package/README.zh.md +4 -4
  6. package/assets/claude/commands/forge-loop.md +7 -1
  7. package/assets/claude/commands/ship.md +18 -1
  8. package/assets/opencode/forgen.ts +52 -0
  9. package/dist/checks/_shared/meta-guard-dispatch.d.ts +7 -0
  10. package/dist/checks/_shared/meta-guard-dispatch.js +12 -2
  11. package/dist/checks/_shared/model-profile.d.ts +25 -0
  12. package/dist/checks/_shared/model-profile.js +63 -0
  13. package/dist/cli.js +99 -153
  14. package/dist/core/auto-compound-runner.d.ts +0 -11
  15. package/dist/core/auto-compound-runner.js +127 -79
  16. package/dist/core/compound-consent.d.ts +15 -0
  17. package/dist/core/compound-consent.js +48 -0
  18. package/dist/core/compound-sweep-cli.d.ts +57 -0
  19. package/dist/core/compound-sweep-cli.js +354 -0
  20. package/dist/core/config-injector.d.ts +16 -1
  21. package/dist/core/config-injector.js +47 -28
  22. package/dist/core/dashboard-cli.js +40 -16
  23. package/dist/core/dashboard.d.ts +3 -4
  24. package/dist/core/dashboard.js +15 -34
  25. package/dist/core/dev-cli.d.ts +13 -0
  26. package/dist/core/dev-cli.js +70 -0
  27. package/dist/core/doctor.d.ts +16 -5
  28. package/dist/core/doctor.js +161 -176
  29. package/dist/core/drift-score.d.ts +2 -0
  30. package/dist/core/drift-score.js +9 -1
  31. package/dist/core/harness.js +115 -43
  32. package/dist/core/health-cli.d.ts +2 -0
  33. package/dist/core/health-cli.js +6 -1
  34. package/dist/core/host-detect.d.ts +3 -1
  35. package/dist/core/host-detect.js +26 -1
  36. package/dist/core/migrate-cli.js +15 -0
  37. package/dist/core/migrate-evidence-host.d.ts +2 -1
  38. package/dist/core/migrate-tenetx.d.ts +50 -0
  39. package/dist/core/migrate-tenetx.js +262 -0
  40. package/dist/core/probe-workflow-cli.d.ts +3 -3
  41. package/dist/core/probe-workflow-cli.js +13 -13
  42. package/dist/core/recall-cli.js +1 -1
  43. package/dist/core/regress-map-cli.js +1 -1
  44. package/dist/core/rendered-rules-manifest.d.ts +29 -0
  45. package/dist/core/rendered-rules-manifest.js +60 -0
  46. package/dist/core/session-store.js +14 -3
  47. package/dist/core/settings-injector.d.ts +3 -0
  48. package/dist/core/settings-injector.js +12 -16
  49. package/dist/core/spawn.d.ts +11 -1
  50. package/dist/core/spawn.js +79 -7
  51. package/dist/core/state-gc.js +1 -0
  52. package/dist/core/status-cli.d.ts +20 -0
  53. package/dist/core/status-cli.js +100 -0
  54. package/dist/core/statusline-cli.d.ts +7 -0
  55. package/dist/core/statusline-cli.js +63 -19
  56. package/dist/core/transcript-summary.d.ts +18 -0
  57. package/dist/core/transcript-summary.js +81 -0
  58. package/dist/core/trust-layer-intent.d.ts +21 -1
  59. package/dist/core/trust-layer-intent.js +7 -0
  60. package/dist/core/types.d.ts +3 -2
  61. package/dist/core/uninstall.js +12 -0
  62. package/dist/core/usage-telemetry.d.ts +7 -1
  63. package/dist/core/usage-telemetry.js +7 -9
  64. package/dist/core/v1-bootstrap.js +1 -1
  65. package/dist/core/watch-cli.js +1 -1
  66. package/dist/engine/compound-extractor.js +10 -0
  67. package/dist/engine/compound-loop.js +35 -6
  68. package/dist/engine/compound-share.d.ts +85 -0
  69. package/dist/engine/compound-share.js +606 -0
  70. package/dist/engine/correction-cluster-runner.d.ts +38 -0
  71. package/dist/engine/correction-cluster-runner.js +188 -0
  72. package/dist/engine/correction-clustering.d.ts +80 -0
  73. package/dist/engine/correction-clustering.js +167 -0
  74. package/dist/engine/enforce-classifier.d.ts +10 -1
  75. package/dist/engine/enforce-classifier.js +111 -22
  76. package/dist/engine/extraction-session.js +10 -3
  77. package/dist/engine/private-filter.d.ts +36 -0
  78. package/dist/engine/private-filter.js +100 -0
  79. package/dist/engine/ranking-pipeline.js +4 -2
  80. package/dist/engine/relevance-gate.d.ts +12 -0
  81. package/dist/engine/relevance-gate.js +12 -0
  82. package/dist/engine/roi-demotion.d.ts +79 -0
  83. package/dist/engine/roi-demotion.js +159 -0
  84. package/dist/engine/solution-format.d.ts +1 -0
  85. package/dist/engine/solution-format.js +26 -0
  86. package/dist/engine/solution-matcher.js +10 -1
  87. package/dist/fgx.js +7 -6
  88. package/dist/forge/cli.js +8 -2
  89. package/dist/hooks/compound-reflection.js +6 -1
  90. package/dist/hooks/context-guard.d.ts +16 -1
  91. package/dist/hooks/context-guard.js +92 -46
  92. package/dist/hooks/post-tool-use.js +2 -3
  93. package/dist/hooks/pre-compact.js +14 -0
  94. package/dist/hooks/pre-tool-use.js +5 -1
  95. package/dist/hooks/shared/stop-triggers.d.ts +29 -2
  96. package/dist/hooks/shared/stop-triggers.js +35 -2
  97. package/dist/hooks/solution-injector.d.ts +28 -0
  98. package/dist/hooks/solution-injector.js +126 -28
  99. package/dist/hooks/stop-guard.js +5 -2
  100. package/dist/hooks/subagent-stop-guard.js +4 -1
  101. package/dist/host/capabilities-claude.js +1 -0
  102. package/dist/host/capabilities-codex.js +1 -0
  103. package/dist/host/capabilities-opencode.d.ts +26 -0
  104. package/dist/host/capabilities-opencode.js +78 -0
  105. package/dist/host/capabilities-registry.d.ts +7 -0
  106. package/dist/host/capabilities-registry.js +14 -0
  107. package/dist/host/exec-host.d.ts +4 -3
  108. package/dist/host/exec-host.js +2 -0
  109. package/dist/host/host-binding.d.ts +27 -0
  110. package/dist/host/host-binding.js +11 -0
  111. package/dist/host/host-runtime.js +18 -0
  112. package/dist/host/install-codex.d.ts +8 -0
  113. package/dist/host/install-codex.js +2 -2
  114. package/dist/host/install-opencode.d.ts +38 -0
  115. package/dist/host/install-opencode.js +148 -0
  116. package/dist/host/install-orchestrator.d.ts +4 -1
  117. package/dist/host/install-orchestrator.js +20 -1
  118. package/dist/host/invoke-agent.d.ts +3 -2
  119. package/dist/host/opencode/context-cli.d.ts +15 -0
  120. package/dist/host/opencode/context-cli.js +25 -0
  121. package/dist/host/opencode/guard-cli.d.ts +13 -0
  122. package/dist/host/opencode/guard-cli.js +39 -0
  123. package/dist/host/opencode/plugin/forgen.d.ts +31 -0
  124. package/dist/host/opencode/plugin/forgen.js +60 -0
  125. package/dist/host/opencode/translate.d.ts +49 -0
  126. package/dist/host/opencode/translate.js +96 -0
  127. package/dist/host/parity-harness.d.ts +10 -2
  128. package/dist/host/projection.js +11 -0
  129. package/dist/mcp/tools.js +17 -4
  130. package/dist/store/evidence-store.d.ts +2 -6
  131. package/dist/store/evidence-store.js +68 -14
  132. package/dist/store/host-mismatch.d.ts +2 -1
  133. package/dist/store/host-mismatch.js +8 -8
  134. package/dist/store/profile-store.d.ts +3 -2
  135. package/dist/store/types.d.ts +9 -2
  136. package/package.json +3 -2
  137. package/plugin.json +2 -2
  138. package/skills/forge-loop/SKILL.md +7 -1
  139. package/skills/ship/SKILL.md +18 -1
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://claude.ai/schemas/claude-plugin.json",
3
3
  "name": "forgen",
4
- "version": "0.4.13",
4
+ "version": "0.5.0",
5
5
  "description": "Claude Code harness — the more you use Claude, the better it gets",
6
6
  "author": {
7
7
  "name": "jang-ujin",
package/CHANGELOG.md CHANGED
@@ -7,6 +7,100 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.5.0] — 2026-08-05 — cross-session δ 실증 · 학습 아키텍처 완성 (ADR-010~013)
11
+
12
+ forgen 의 첫 npm 릴리스(이전 발행은 0.4.x). 헤드라인: **forgen 효과(δ)를 프론티어
13
+ 모델에서 처음으로 유의하게 실증**했고, 그 학습을 안전·정직하게 돌리는 백그라운드
14
+ 아키텍처(ADR-011~013)를 완성했다. 아래 플랫폼-수렴 재정렬(ADR-010)이 이 릴리스의 토대다.
15
+
16
+ ### 측정 — cross-session δ 실증 (핵심)
17
+ - **forgen 효과(δ)를 처음으로 유의하게 실증** — cross-session 정책 준수. forgen 이
18
+ 이전 세션의 user-특정 정책을 재주입해, 그 정책을 모르는 vanilla 대비 준수율을 올린다:
19
+ Sonnet 5 **δ = 0.714, 95% CI [0.429, 0.929]** (vanilla 4/14 → forgen 14/14),
20
+ Opus 4.8 **δ = 0.500, 95% CI [0.214, 0.786]** (vanilla 7/14 → forgen 14/14).
21
+ 블라인드 human rater 확증(Cohen's κ = 0.748). **"내부 측정" 등급** (메인테이너 유래
22
+ 데이터셋 + intra-family judge + rater 1인) — 외부 주장은 독립 rater 패널·외부 리뷰어·
23
+ cross-family judge 재현 후. `docs/release/v0.5.0-persistence-delta.md`.
24
+ - **δ 는 모델 역량과 직교**: 강한 모델일수록 일반 good-practice 는 스스로 따라 δ 가
25
+ 작아지나, user-특정(추론 불가) 정책은 forgen 만 맞힘. within-session 은 구조적 null
26
+ (프론티어 컨텍스트 보유) — δ 는 cross-session 에서만 산다.
27
+
28
+ ### Added — 학습 백그라운드 아키텍처
29
+ - **auto-compound 백그라운드 방법론 (ADR-011)**: `forgen compound sweep` 시간-기반
30
+ backstop + cron 자동설치(`--install-cron`), barren backoff 성장예외, PreCompact 러너.
31
+ 긴 컴팩션/차단 세션에서 학습이 유실되던 갭 해소.
32
+ - **auto-compound 동의 모델 (ADR-012)**: transcript 요약을 Haiku 로 전송하는 추출을
33
+ **opt-in** 화(`forgen compound consent on|off`). 결정론 교정→룰 승급(egress 0)은 기본
34
+ 유지 — "당신 몰래 API 로 보내지 않는다".
35
+ - **correction-aware 채굴 (ADR-013)**: 명시 마커 없는 완곡·반사실 교정을 사후 회수해
36
+ 학습에 반영. 채굴 룰은 **advisory-only invariant**(차단 불가) + 캡 + auto: 네임스페이스
37
+ + 생성나이 은퇴 — LLM 채굴이 위험한 영구 차단 룰을 만들지 못하게 data-level 로 보장.
38
+
39
+ ### Fixed — 학습 파이프라인 실버그
40
+ - **실 transcript 스키마 버그**: auto-compound-runner 가 text 를 top-level `entry.content`
41
+ 로 읽었으나 실제 Claude 스키마는 `entry.message.content` 중첩 → 전 user/assistant 턴
42
+ 누락으로 auto-compound(채굴 포함)가 **실데이터에서 dead** 였다. message.content 우선 +
43
+ fallback, 순수 파싱 모듈 분리 + real-schema 회귀.
44
+ - **settings-injection 마커 버그**: `permissions.deny` JSON 배열에 `# forgen-managed`
45
+ 주석을 데이터로 주입 → Claude Code 가 malformed 룰로 매 세션 경고. 주입 중단 +
46
+ 기존 오염 self-heal.
47
+ - **drift-score hardcap cooldown 부재**: 50편집 초과 후 매 편집마다 drift_critical 이
48
+ 발화하던(세션길이 카운터로 전락) 문제.
49
+
50
+ ---
51
+
52
+ ### 플랫폼 수렴 대응 (ADR-010, 이 릴리스의 토대 — 2026-07-16)
53
+
54
+ Claude Code 가 forgen 영역을 native 로 흡수하기 시작한 것(`/doctor`, `/usage`,
55
+ Auto mode)과 Sonnet 5 기본 모델 전환에 대한 전략 릴리스. 방향: native 와 겹치는
56
+ 표면에서 물러나고, moat(교정→프로필 학습·증거 게이팅 정책·compound recall+ROI·
57
+ multi-host)에 재집중. 상세 결정 기록: `docs/adr/ADR-010`, 실행 스펙:
58
+ `docs/plans/2026-07-16-v0.5.0-execution-plan.md`. 전 작업 청크가 적대적 리뷰
59
+ 6회전을 거침 (SEV-1 3건 포함 전 발견 반영).
60
+
61
+ #### 측정 기반 재정렬 (핵심 배경)
62
+ - **v0.4.11 실측: opus-4.8 에서 완료 가드 blocks=0** (easy/hard 양쪽) — δ 효과는
63
+ 100% injection 에서 나온다. enforcement 는 프론티어 모델이 흡수했고, forgen 의
64
+ 가치는 injection 품질·메모리·개인화·측정으로 이동했다. 효과 수치는 Sonnet 5
65
+ 재측정(R2) 전까지 주장하지 않는다 (Honest Fail Path).
66
+
67
+ #### Added
68
+ - **`forgen migrate tenetx`** (+ `doctor --reclaim`): tenetx(레거시 정체성)/구버전
69
+ forgen 이 `~/.claude/rules/` 등에 남긴 규칙 스프롤을 provenance 기반으로 회수.
70
+ manifest content-hash 일치 = 무프롬프트, 마커만 = `--yes` 필요, 그 외 불간섭.
71
+ 전량 백업-이동(가역), `--dry-run`, `--apply-settings`(settings-lock 준수).
72
+ - **Injection ROI 루프**: `surfaced ≫ acted_on` 저 ROI 솔루션 자동 강등(×0.5,
73
+ 24h 게이트 2윈도 연속 시 주입 제외, acted 발생 시 즉시 해제). `forgen status`
74
+ 에 "Surfaced but ignored" 패널. native 메모리에 없는 acted-on 피드백 루프.
75
+ - **per-model 완료-가드 프로필**: 측정된 opus-4.8 은 advise(기록만), 미측정
76
+ 모델(sonnet-5 포함)은 block 유지. 모델 식별은 statusline 세션 캐시 경유
77
+ (hook stdin 에 모델 필드 없음 — probe 실측). DANGEROUS 가드는 모델 무관 block.
78
+ - **smoke 릴리스 게이트**: `scripts/smoke.cjs` — 실제 프로세스 실행 증거만 기록
79
+ (vitest/cli/statusline/hook-exec), report.version ↔ package.json 바인딩으로
80
+ stale 증거 재사용 차단. Docker e2e 게이트 폐지 대체 (증거 수단 적정화 —
81
+ 원칙 유지).
82
+ - retro-real 평가 데이터셋 15엔트리 (실세션 anonymized, 완화 역방향 케이스 포함).
83
+
84
+ #### Changed
85
+ - **경계 재정의**: `forgen doctor` 는 forgen 자체 기계+효과-측정 게이트만
86
+ (환경 건강 → native `/doctor` 안내, Maturity/QuickWins 제거, hook timing →
87
+ `--verbose`); 사용량 표시 → native `/usage` 이관(1회 공지, dashboard 는
88
+ deprecated 플래그); 보안/안티패턴 prose → 훅 포인터 2줄 (71.8% 축소).
89
+ - **규칙 주입 전면 프로젝트 스코프화**: `forge-*` 글로벌 사이드채널 제거 —
90
+ behavioral 패턴이 전 프로젝트에 새던 결함(F6) 수정. 렌더 파일 content-hash
91
+ manifest 기록 (reclaimer 근거).
92
+ - **behavioral 에코 하드닝**: observedCount≥2 게이트(주력 — 실오염 49/49 가
93
+ 1회 관찰) + 캡처 사이드 차단 + 앵커드 패턴 보강. Claude-voice 에코가 규칙으로
94
+ 재주입되던 오염 경로 차단.
95
+
96
+ #### Fixed
97
+ - `doctor --repair` 가 글로벌 설치에서 실제로 복구하지 못하던 버그 (devDeps 부재로
98
+ build 실패 → postinstall 미도달) — dist 존재 시 build 생략 + 결과 재검증 보고.
99
+ - statusline 1회 공지가 5초 캐시에 섞여 반복되던 버그.
100
+
101
+ #### Deprecated
102
+ - `usage-telemetry` 기록 (no-op shim, v0.6.0 모듈 삭제 예정) — native `/usage`.
103
+
10
104
  ## [0.4.13] — 2026-06-02 — Hotfix: fgx 세션 종료 시 터미널 물림 방지
11
105
 
12
106
  핫픽스 한 건.
package/README.ja.md CHANGED
@@ -55,19 +55,19 @@ Claude: 「完了宣言を取り消します。証拠ファイルがありま
55
55
 
56
56
  **何が起きたか**: Claude の Stop フックが、あなたが定義したルール (`L1-e2e-before-done`) によってブロックされました。Claude はブロック `reason` を読み、早すぎた完了宣言を撤回し、証拠を生成して再提出しました。**追加 API コール 0** — Claude が元々生成する予定だった同じセッションのターン内で全て起きました。
57
57
 
58
- これが **Mech-B セルフチェックプロンプトインジェクト** です。Claude Code の Stop フックが `decision: "block"` + `reason` を受け入れ、Claude が次のターンでその reason を入力として読むから成り立ちます。10 シナリオ $1.74 で end-to-end 検証済み ([A1 spike report](docs/spike/mech-b-a1-verification-report.md))。
58
+ これが **Mech-B セルフチェックプロンプトインジェクト** です。Claude Code の Stop フックが `decision: "block"` + `reason` を受け入れ、Claude が次のターンでその reason を入力として読むから成り立ちます。10 シナリオ $1.74 で end-to-end 検証済み (A1 spike report — git 履歴にアーカイブ済み, `docs/spike/` pre-v0.5.0)。
59
59
 
60
60
  🎬 **動作を見る** (27秒):
61
61
 
62
62
  ```bash
63
63
  # 実際のフック・実際のルール・実際の block/approve サイクルをライブで
64
- bash docs/demo/mech-b-demo.sh
64
+ # demo スクリプトは git 履歴にアーカイブ済み (docs/demo/ pre-v0.5.0)
65
65
 
66
66
  # または録画済み asciinema キャストを再生
67
- asciinema play docs/demo/mech-b-block-unblock.cast
67
+ # asciinema cast も同様
68
68
  ```
69
69
 
70
- デモの「本物/シミュレーション」の区別は [`docs/demo/README.md`](docs/demo/README.md) を参照。
70
+
71
71
 
72
72
  ---
73
73
 
package/README.md CHANGED
@@ -3,8 +3,9 @@
3
3
  </p>
4
4
 
5
5
  <p align="center">
6
- <strong>When your agent says "done", forgen makes it prove it.</strong><br/>
7
- Turn-level self-verification + personalized rules for <strong>Claude Code</strong> and <strong>Codex CLI</strong>, at <strong>$0 extra API cost</strong>.
6
+ <strong>forgen makes Claude fit <em>you</em> — and measures whether that actually helps, honestly.</strong><br/>
7
+ Your corrections become persistent rules; past solutions resurface when relevant; stale ones fade.<br/>
8
+ For <strong>Claude Code</strong> and <strong>Codex CLI</strong>, at <strong>$0 extra API cost</strong>.
8
9
  </p>
9
10
 
10
11
  <p align="center">
@@ -35,7 +36,7 @@
35
36
 
36
37
  ## The first block (30 seconds)
37
38
 
38
- You've been burned: Claude says "tests pass, implementation done" — you run it — it doesn't work. forgen closes that gap.
39
+ Here's forgen's most *visible* mechanism — a live guard you can trigger in 30 seconds. Fair warning up front (this is the honest-measurement tool, after all): on current frontier models this particular guard rarely fires, because Opus 4.8 / Sonnet 5 mostly stopped making unverified "done" claims on their own (we measured **blocks = 0** — see the caveat below). So treat this as a **safety net for weaker models and the occasional tail case**, not forgen's daily value. The daily value is quieter: your corrections becoming rules, and past solutions resurfacing. But the guard is the fastest thing to *see*, so we lead with it.
39
40
 
40
41
  ```
41
42
  You: "Update the auth middleware."
@@ -57,7 +58,7 @@ Claude: "측정 없이 점수를 매겼습니다. 실 테스트부터 실행합
57
58
 
58
59
  The same mechanism also fires when Claude writes conclusions faster than evidence ("done. passed. shipped. verified." with no measurement context), or claims facts ("테스트가 통과합니다") without ever having executed them. You can also define **custom rules** (e.g. "require npm test evidence before saying 'done' in this repo") via `forgen compound --rule` — they slot into the same Stop-hook dispatcher.
59
60
 
60
- This is **Mech-B self-check prompt-inject**. It works because Claude Code's Stop hook accepts `decision: "block"` + `reason`, and Claude in the next turn reads that reason as input. Codex CLI gets the same treatment via the symmetric host adapter (v0.4.3, [multi-host core design](docs/superpowers/specs/2026-04-27-forgen-multi-host-core-design.md)). We verified it end-to-end on 10 scenarios at $1.74 total cost ([A1 spike report](docs/spike/mech-b-a1-verification-report.md)), and v0.4.1 added built-in guards so you get the first block **without writing any rule**.
61
+ This is **Mech-B self-check prompt-inject**. It works because Claude Code's Stop hook accepts `decision: "block"` + `reason`, and Claude in the next turn reads that reason as input. Codex CLI gets the same treatment via the symmetric host adapter (v0.4.3, [multi-host core design](docs/superpowers/specs/2026-04-27-forgen-multi-host-core-design.md)). We verified it end-to-end on 10 scenarios at $1.74 total cost (A1 spike report — archived in git history, `docs/spike/` pre-v0.5.0), and v0.4.1 added built-in guards so you get the first block **without writing any rule**.
61
62
 
62
63
  > **v0.4.3 self-correction story:** the same guards detected their own 16-day false-positive (strict φ 65.66% — 84% from a single Korean-regex bug), and the [`forgen-eval`](packages/forgen-eval/) introspect testbed (alpha) flagged a `TEST-1` wiring gap on top of it. Both fixes shipped in v0.4.3 — forgen finding and fixing forgen. Details in [CHANGELOG](CHANGELOG.md).
63
64
 
@@ -73,18 +74,53 @@ This is **Mech-B self-check prompt-inject**. It works because Claude Code's Stop
73
74
  > is rescinded as a broken-testbed artifact; ψ (forgen+mem coexistence) is
74
75
  > confirmed as ≈0 — **forgen alone is the recommended path**. See
75
76
  > [`docs/release/v0.4.5-draft.md`](docs/release/v0.4.5-draft.md).
76
-
77
- 🎬 **See it happen** (27 seconds):
77
+ >
78
+ > **Scope caveat (v0.5.0, ADR-010 + 2026-07-20 honest null):** the δ numbers
79
+ > above were measured on the **models of their era** (Claude Sonnet 4.x-class,
80
+ > Codex) and we do **not** carry them forward as a claim about current frontier
81
+ > models. Two separate current-model findings, kept distinct:
82
+ >
83
+ > 1. **Completion-guard / false-completion pressure → null (honest).** On
84
+ > **Opus 4.8** and **Sonnet 5** we measured completion-guard **blocks = 0** —
85
+ > frontier models got honest on their own. Our v0.5.0 judge redesign re-scored the
86
+ > hard "false-completion pressure" cases and found **no measurable behavioral δ
87
+ > on that dataset** (all arms avoid blatant false completion;
88
+ > [`docs/release/v0.5.0-recalibration.md`](docs/release/v0.5.0-recalibration.md)).
89
+ > We publish that null rather than bury it.
90
+ >
91
+ > 2. **Cross-session user-specific policy adherence → δ > 0 (measured).** forgen's
92
+ > real mechanism is not blocking but *re-injecting your corrections in a later
93
+ > session where a vanilla model has no memory of them*. On that — the honest place
94
+ > the effect lives — we measured a **significant δ**: forgen makes the model follow
95
+ > the user's prior-session policies **14/14**, vanilla **4/14 (Sonnet 5)** / **7/14
96
+ > (Opus 4.8)** → **δ = 0.714, 95% CI [0.429, 0.929]** (Sonnet 5) and **δ = 0.500,
97
+ > 95% CI [0.214, 0.786]** (Opus 4.8), confirmed by a blind human rater (Cohen's
98
+ > κ = 0.748 vs the LLM judge)
99
+ > ([`docs/release/v0.5.0-persistence-delta.md`](docs/release/v0.5.0-persistence-delta.md)).
100
+ > **δ is orthogonal to model capability** — the stronger model (Opus) follows more
101
+ > general good-practice on its own, so δ shrinks there, but on *user-specific*
102
+ > policies neither model guesses right and only forgen does. This is
103
+ > **"internal measurement" grade**: the dataset is maintainer-authored (bootstrap),
104
+ > the judge is intra-family (Claude), and only one human rater has confirmed it. We
105
+ > do **not** make external effect claims until an independent rater panel + external
106
+ > reviewer + a cross-family judge reproduce it.
107
+ >
108
+ > What forgen reliably provides: **personalization** (corrections → persistent
109
+ > rules, cross-session δ above), **recall** (past solutions injected when relevant),
110
+ > **ROI demotion** of stale knowledge, and **deterministic guards** (secrets,
111
+ > destructive commands) that fire regardless of model honesty. Per-model guard
112
+ > behavior: measured-honest models get advisory mode instead of blocking.
113
+
114
+ 🎬 **See it happen** — try the guard live on your own install:
78
115
 
79
116
  ```bash
80
- # Watch the full loop live — actual hook, actual rule, actual block/approve cycle
81
- bash docs/demo/mech-b-demo.sh
82
-
83
- # Or replay the pre-recorded asciinema cast
84
- asciinema play docs/demo/mech-b-block-unblock.cast
117
+ # Real hook, real block: a self-score claim with zero measurement evidence
118
+ echo '{"session_id":"demo","last_assistant_message":"이번 작업 신뢰도 90% 로 평가됩니다."}' \
119
+ | node "$(npm root -g)/@wooojin/forgen/dist/hooks/stop-guard.js"
120
+ # → decision:"block" + the exact reason Claude reads on its next turn
85
121
  ```
86
122
 
87
- See [`docs/demo/README.md`](docs/demo/README.md) for what's real vs simulated in the demo.
123
+ (The pre-v0.5.0 scripted demo lived in `docs/demo/` — archived in git history.)
88
124
 
89
125
  ---
90
126
 
@@ -179,20 +215,21 @@ Behind the scenes:
179
215
 
180
216
  You say: "Don't refactor files I didn't ask you to touch."
181
217
 
182
- Claude calls the `correction-record` MCP tool. The correction is stored as structured evidence with axis classification (`judgment_philosophy`), kind (`avoid-this`), and confidence score. A temporary rule is created for immediate effect in the current session.
218
+ Claude calls the `correction-record` MCP tool. The correction is stored as structured evidence with axis classification (`judgment_philosophy`), kind (`avoid-this`), and confidence score. A temporary rule is created for immediate effect in the current session. At session end it promotes to a **permanent rule** — this deterministic corrections → rules path runs with **zero network egress** and is what our measured cross-session δ rides on.
219
+
220
+ Softer corrections you never explicitly flag ("음… 이렇게 최소로만 짜면 나중에 다 다시 해야 하는 거 아니야?") are harder to catch in real time. If you opt into transcript extraction (below), forgen also **retroactively mines** those from the conversation — but such auto-mined rules are **advisory-only** (they surface as context, never block or enforce) and age out if you never act on them.
183
221
 
184
222
  ### Between sessions (automatic)
185
223
 
186
- When a session ends, auto-compound extracts:
187
- - Solutions (reusable patterns with context)
188
- - Behavioral observations (how you work)
189
- - A session learning summary
224
+ **Always on (deterministic, zero egress):** your explicit corrections promote to permanent rules; near-duplicate corrections cluster and strengthen; facets micro-adjust from accumulated evidence; a time-based `compound sweep` backstop (optionally on cron) makes sure long or interrupted sessions don't lose that learning.
190
225
 
191
- Facets are micro-adjusted based on accumulated evidence. If your corrections consistently point away from your current pack, mismatch detection triggers after 3 sessions and recommends a pack change.
226
+ **Opt-in (`forgen compound consent on`):** a background Haiku pass reads a *redacted summary* of the session transcript to extract reusable solutions, behavioral observations, and the softer corrections above. This is **off by default** — forgen does not send your conversations anywhere unless you turn it on, and secrets / `<private>` ranges are stripped before anything is sent. `forgen doctor` always shows the current state.
227
+
228
+ If your corrections consistently point away from your current pack, mismatch detection triggers after 3 sessions and recommends a pack change.
192
229
 
193
230
  ### Next session
194
231
 
195
- Updated rules are rendered with your corrections included. Compound knowledge is searchable via MCP. Retrieval precision grows as your personal accumulation grows — the mechanism is in place from day 1 (starter-pack covers common dev queries on a fresh install), and the signal-to-noise ratio improves over roughly 2–4 weeks of real use as low-fitness solutions are auto-demoted and your specific patterns get promoted.
232
+ Updated rules are rendered with your corrections included. Compound knowledge is searchable via MCP. The retrieval mechanism is in place from day 1 (starter-pack covers common dev queries on a fresh install); as you accumulate your own solutions, low-fitness ones get auto-demoted (ROI loop) and your specific patterns get promoted, so what surfaces skews toward *your* corpus over time. We describe this as a mechanism, not a measured speedup — we make no effect-size claim (see the honest-null caveat above).
196
233
 
197
234
  ---
198
235
 
@@ -229,7 +266,7 @@ forgen doctor --quick
229
266
 
230
267
  > **Vendor dependency:** Forgen wraps Claude Code and Codex CLI symmetrically (Claude is the behavior reference; Codex extends with equivalence). Upstream API/CLI changes may affect behavior. Tested with Claude Code 1.0.x / 2.1.x and Codex 0.x.
231
268
 
232
- > **Upgrading from v0.4.x?** Run `forgen install claude` (or `codex` / `both`) after upgrading — v0.4.3+ requires explicit host registration. Then `forgen doctor --quick` to verify.
269
+ > **Upgrading from v0.4.x?** Run `forgen install claude` (or `codex` / `both`) after upgrading — v0.4.3+ requires explicit host registration. Then run `forgen migrate tenetx --dry-run` to preview and reclaim legacy global rule sprawl (v0.5.0 moved ALL rule injection to project scope — stale `~/.claude/rules/forge-*.md` from older versions should be reclaimed; see CHANGELOG). Finally `forgen doctor --quick` to verify.
233
270
 
234
271
  ### Try your first block
235
272
 
@@ -349,8 +386,8 @@ entries in `~/.forgen/state/implicit-feedback.jsonl`. Idempotent — safe to re-
349
386
  | |
350
387
  v |
351
388
  +------------------+ |
352
- | Session Ends | auto-compound extracts: |
353
- | | solutions + observations + summary |
389
+ | Session Ends | corrections -> rules (always, egress 0) |
390
+ | | transcript extract: opt-in (Haiku) |
354
391
  +--------+---------+ |
355
392
  | |
356
393
  v |
@@ -399,6 +436,30 @@ Each solution starts as an `experiment`. As it gets reflected in your code acros
399
436
  | **Behavioral patterns** | Auto-detected at 3+ observations | Applied to `forge-behavioral.md` |
400
437
  | **Evidence** | Corrections + observations | Drives facet adjustments + rule creation |
401
438
 
439
+ *Solutions and behavioral patterns come from the **opt-in** transcript-extraction pass (`forgen compound consent on`, off by default). Corrections → rules and facet adjustments are always-on and never leave your machine.*
440
+
441
+ ### Keeping things out of the corpus (`<private>`)
442
+
443
+ Wrap anything you don't want learned in a `<private>…</private>` block, or end a
444
+ line with a `// forgen:private` marker (`#` and `/*` comment styles also work):
445
+
446
+ ```
447
+ Here's the approach — <private>internal API key rotation playbook</private> — use it.
448
+ const token = "…"; // forgen:private
449
+ ```
450
+
451
+ Those ranges are stripped before every learning-corpus capture path: correction
452
+ records, solution extraction, the automatic session-end compound runner,
453
+ prompt-history, and session-search indexing. Unclosed or malformed tags **fail
454
+ closed** (treated as private to end-of-text) so a forgotten `</private>` never
455
+ leaks.
456
+
457
+ > **Scope:** `<private>` controls **what forgen learns from**, not what the model
458
+ > sees. The text was already sent to Claude as part of your conversation — the tag
459
+ > excludes it from persistent capture (compound solutions, evidence, search index),
460
+ > it does not make the current turn private. This is a separate axis from the
461
+ > secret-filter, which blocks credentials from being committed.
462
+
402
463
  ### Solution auto-injection
403
464
 
404
465
  Every prompt you type is matched against your accumulated solutions. Relevant ones are automatically injected into Claude's context — no manual lookup needed.
@@ -672,9 +733,10 @@ forgen skill list # List promoted skills
672
733
 
673
734
  ```bash
674
735
  forgen init # Initialize project (+ 15 starter-pack solutions)
675
- forgen migrate [implicit-feedback|all]
676
- # One-shot schema migrations (idempotent)
677
- forgen doctor # System diagnostics (10 categories + harness maturity)
736
+ forgen migrate [implicit-feedback|evidence-host|tenetx|all]
737
+ # One-shot migrations (idempotent). tenetx: reclaim
738
+ # legacy global rules (--dry-run/--yes/--apply-settings)
739
+ forgen doctor # Forgen-specific diagnostics (plugin cache, hooks, state, parity, gates — env health: native /doctor)
678
740
  forgen doctor --prune-state # Daily hygiene: state GC + T4 rule decay (90d idle → retire)
679
741
  forgen dashboard # Knowledge overview (6 sections)
680
742
  forgen config hooks # View hook status + context budget
@@ -712,7 +774,7 @@ Three CI gates prove forgen does not violate its own L1 rules before release:
712
774
  ```bash
713
775
  node scripts/self-gate.cjs # Static: mock-in-prod, secrets, enforce_via, release-artifact
714
776
  node scripts/self-gate-runtime.cjs # Runtime smoke: 6 hook scenarios
715
- node scripts/self-gate-release.cjs # Tag-only: version/tag/CHANGELOG/dist/e2e-report consistency
777
+ node scripts/self-gate-release.cjs # Tag-only: version/tag/CHANGELOG/dist/smoke-report consistency
716
778
  ```
717
779
 
718
780
  Triggered by `.github/workflows/self-gate.yml` on push main / PR main / tag v*. Dogfood opt-in: see [.forgen/README.md](.forgen/README.md).
@@ -843,7 +905,7 @@ forgen and [claude-mem](https://github.com/thedotmack/claude-mem) solve **comple
843
905
  | **Trigger** | Stop / PreToolUse hooks | UserPromptSubmit hook |
844
906
  | **Cost** | $0 (in-turn block/reason) | $0 (vector recall, local) |
845
907
 
846
- Install both as separate Claude Code plugins (Plugin model — forgen does not bundle claude-mem; AGPL-3.0 stays at arm's length). When both are present forgen's auto-detect yields context budget so claude-mem's recall has room to land, and the orchestration contract — order, failure isolation, Stop-hook ownership — is documented in [ADR-004](docs/adr/ADR-004-claude-mem-hook-orchestration.md). The pairing is one of the 5 arms tracked by [forgen-eval](packages/forgen-eval/) (see [claude-mem spike](docs/spike/2026-04-28-claude-mem-spike.md)).
908
+ Install both as separate Claude Code plugins (Plugin model — forgen does not bundle claude-mem; AGPL-3.0 stays at arm's length). When both are present forgen's auto-detect yields context budget so claude-mem's recall has room to land, and the orchestration contract — order, failure isolation, Stop-hook ownership — is documented in [ADR-004](docs/adr/ADR-004-claude-mem-hook-orchestration.md). The pairing is one of the 5 arms tracked by [forgen-eval](packages/forgen-eval/) (claude-mem spike report — archived in git history, `docs/spike/` pre-v0.5.0).
847
909
 
848
910
  ```
849
911
  You: "fix the auth flow"
package/README.zh.md CHANGED
@@ -55,19 +55,19 @@ Claude: "撤回完成声明。证据文件不存在。先执行 e2e..."
55
55
 
56
56
  **刚刚发生了什么**: Claude 的 Stop hook 被你定义的规则 (`L1-e2e-before-done`) 拦截。Claude 读取了 block `reason`, 撤回过早的完成声明, 产生证据, 重新提交。**零额外 API 调用** — 全部发生在 Claude 本来就会产出的同一个 session turn 内。
57
57
 
58
- 这就是 **Mech-B 自检 prompt-inject**。它工作是因为 Claude Code 的 Stop hook 接受 `decision: "block"` + `reason`, 而 Claude 在下一轮把那个 reason 作为输入读取。我们用 10 个场景、$1.74 总成本端到端验证 ([A1 spike report](docs/spike/mech-b-a1-verification-report.md))。
58
+ 这就是 **Mech-B 自检 prompt-inject**。它工作是因为 Claude Code 的 Stop hook 接受 `decision: "block"` + `reason`, 而 Claude 在下一轮把那个 reason 作为输入读取。我们用 10 个场景、$1.74 总成本端到端验证 (A1 spike report — 已归档于 git 历史, `docs/spike/` pre-v0.5.0)。
59
59
 
60
60
  🎬 **观看实际运行** (27秒):
61
61
 
62
62
  ```bash
63
63
  # 现场观看完整循环 — 真实的 hook、真实的规则、真实的 block/approve 周期
64
- bash docs/demo/mech-b-demo.sh
64
+ # demo 脚本已归档于 git 历史 (docs/demo/ pre-v0.5.0)
65
65
 
66
66
  # 或重放预录制的 asciinema cast
67
- asciinema play docs/demo/mech-b-block-unblock.cast
67
+ # asciinema cast 同上
68
68
  ```
69
69
 
70
- 关于 demo 中"真实 vs 模拟"的详情见 [`docs/demo/README.md`](docs/demo/README.md)。
70
+
71
71
 
72
72
  ---
73
73
 
@@ -88,13 +88,19 @@ PRD 확정 직후 **반드시** `~/.forgen/state/forge-loop.json`에 저장:
88
88
  mkdir -p ~/.forgen/state
89
89
  cat > ~/.forgen/state/forge-loop.json <<EOF
90
90
  {"active":true,"startedAt":"$(date -u +%Y-%m-%dT%H:%M:%SZ)","stories":[
91
- {"id":"US-001","title":"...","passes":false,"attempts":0}
91
+ {"id":"US-001","title":"...","passes":false,"attempts":0,"acceptanceCriteria":["..."]}
92
92
  ]}
93
93
  EOF
94
94
  ```
95
95
 
96
96
  이 파일이 있어야 Claude가 중간에 멈추지 않도록 Stop 훅이 차단합니다.
97
+ 소유 세션은 최초 차단 시점에 Stop 훅이 자동 귀속하므로(`sessionId` 필드) 이
98
+ 파일을 쓸 때 세션 ID를 직접 넣을 필요는 없습니다 — 단, 귀속된 세션과 다른
99
+ 세션에서는 이 루프가 차단하지 않습니다. 24시간 이상 갱신이 없으면 자동
100
+ 해제(1회성 안내 포함)되며, 연속 30회 차단 시에도 안전 상한으로 자동 해제됩니다.
97
101
  스토리 완료 시 `passes: true`로 업데이트. 전체 완료는 Stop 훅이 자동 처리.
102
+ `acceptanceCriteria[0]`은 차단 메시지에 `AC1:`로 노출되므로 있으면 첫 항목을
103
+ 구체적이고 검증 가능한 문장으로 작성하세요 (없어도 정상 동작).
98
104
 
99
105
  ### goal-only 모드 — Phase 1 종료 분기
100
106
 
@@ -166,10 +166,27 @@ npm version {patch|minor|major} --no-git-tag-version
166
166
 
167
167
  커밋을 주제별 그룹핑 -> CHANGELOG.md 상단에 추가.
168
168
 
169
+ ## Step 6.5: Smoke 증거 생성 (forgen 레포 한정)
170
+
171
+ forgen 레포 자체를 ship 할 때는 릴리스 커밋 전에 smoke 증거를 생성한다
172
+ (ADR-010 W0-2 — CI self-gate 가 릴리스 커밋에서 smoke-report 를 요구).
173
+
174
+ ```bash
175
+ # forgen 레포에서만. vitest 는 Step 3 에서 이미 돌았지만 게이트 증거는
176
+ # smoke.cjs 산출물만 인정 — 전체 실행 (실제 프로세스 산출물 원칙).
177
+ if [ -f scripts/smoke.cjs ]; then
178
+ node scripts/smoke.cjs || ABORT
179
+ fi
180
+ ```
181
+
182
+ - 실패 -> ABORT ("smoke 증거 생성 실패 — 게이트 통과 불가")
183
+
169
184
  ## Step 7: 릴리스 커밋
170
185
 
171
186
  ```bash
172
- git add package.json CHANGELOG.md
187
+ git add package.json package-lock.json CHANGELOG.md
188
+ # forgen 레포: smoke 증거 포함
189
+ [ -f .forgen-release/smoke-report.json ] && git add .forgen-release/smoke-report.json
173
190
  git commit -m "release: v{version}"
174
191
  ```
175
192
 
@@ -0,0 +1,52 @@
1
+ // forgen-managed — do not edit; regenerated by `forgen install opencode`
2
+ /**
3
+ * forgen — OpenCode plugin (W3-3 P1). forgen 결정적 가드 + 개인화 브릿지:
4
+ * - tool.execute.before → forgen opencode-guard → block 이면 throw(도구 차단).
5
+ * - experimental.session.compacting → forgen opencode-context → forge-loop 상태 유지.
6
+ * 얇은 shim — 로직은 forgen 소유(drift-free). async execFile(이벤트루프 비차단).
7
+ * GUARD_CMD/CONTEXT_CMD 는 install-opencode 가 절대경로로 치환(런타임 PATH 비의존).
8
+ */
9
+ import { execFile } from "node:child_process"
10
+
11
+ // forgen:guard-cmd — install-opencode 가 [node, <절대 cli 경로>, opencode-guard] 로 치환
12
+ const GUARD_CMD: string[] = ["forgen", "opencode-guard"]
13
+ // forgen:context-cmd — install-opencode 가 [node, <절대 cli 경로>, opencode-context] 로 치환
14
+ const CONTEXT_CMD: string[] = ["forgen", "opencode-context"]
15
+
16
+ function runForgen(cmd: string[], stdin: string, timeout: number): Promise<string> {
17
+ return new Promise((resolve) => {
18
+ const [bin, ...rest] = cmd
19
+ const child = execFile(bin, rest, { timeout, encoding: "utf-8" }, (err, stdout) => {
20
+ if (err) console.error("[forgen] 호출 실패(fail-open):", err.message)
21
+ resolve(stdout || "")
22
+ })
23
+ child.stdin?.end(stdin)
24
+ })
25
+ }
26
+
27
+ export const forgen = async () => ({
28
+ "tool.execute.before": async (
29
+ input: { tool?: string },
30
+ output: { args?: Record<string, unknown> },
31
+ ) => {
32
+ const out = await runForgen(GUARD_CMD, JSON.stringify({ tool: input?.tool, args: output?.args }), 8000)
33
+ let decision: { block?: boolean; reason?: string } = { block: false }
34
+ try {
35
+ if (out) decision = JSON.parse(out)
36
+ } catch {
37
+ /* fail-open */
38
+ }
39
+ if (decision.block) throw new Error(decision.reason || "[forgen] blocked by guard")
40
+ },
41
+ "experimental.session.compacting": async (
42
+ _input: unknown,
43
+ output: { context?: string[] },
44
+ ) => {
45
+ const ctx = await runForgen(CONTEXT_CMD, "", 5000)
46
+ if (ctx.trim() && Array.isArray(output?.context)) {
47
+ output.context.push(ctx.trim())
48
+ }
49
+ },
50
+ })
51
+
52
+ export default forgen
@@ -17,6 +17,13 @@ export interface MetaGuardContext {
17
17
  recentTools: string[];
18
18
  /** TEST-1 fact-vs-agreement 최소 측정 횟수 (기본 1). */
19
19
  minMeasurements?: number;
20
+ /**
21
+ * W4-3 (ADR-010): 완료 가드(TEST-1/2/3)의 동작 모드. 'advise' 면 block 을
22
+ * correction(기록만)으로 강등한다 — 측정된 프론티어 모델(opus-4.8 blocks=0)
23
+ * 에서 잔여 발화는 거짓양성 개연성이 높으므로. DANGEROUS-RESPONSE 는 모델
24
+ * 무관 안전장치라 이 모드의 영향을 받지 않는다. 기본 'block' (현행 유지).
25
+ */
26
+ completionGuardMode?: 'block' | 'advise';
20
27
  }
21
28
  export interface MetaGuardResult {
22
29
  /** 짧은 식별자 (builtin:<shortId> 형태의 rule_id 와 reason prefix 에 사용). */
@@ -68,13 +68,23 @@ export function runMetaGuards(ctx) {
68
68
  },
69
69
  ];
70
70
  const results = [];
71
+ const adviseMode = ctx.completionGuardMode === 'advise';
71
72
  for (const c of checks) {
72
73
  const out = c.run();
73
74
  if (!out.triggered)
74
75
  continue;
75
- results.push({ shortId: c.shortId, ruleSlug: c.ruleSlug, kind: c.kind, reason: out.reason });
76
+ // W4-3: advise 모드에선 완료 가드(TEST-*)의 block 을 correction 으로 강등.
77
+ // DANGEROUS 는 모델 무관 결정적 안전장치 — 강등 대상 아님.
78
+ const effectiveKind = adviseMode && c.kind === 'block' && c.shortId !== 'dangerous-response-pattern'
79
+ ? 'correction'
80
+ : c.kind;
81
+ results.push({ shortId: c.shortId, ruleSlug: c.ruleSlug, kind: effectiveKind, reason: out.reason });
82
+ // 원래 kind 기준으로 중단 (강등돼도 동일) — 리뷰 SEV-1: 강등 결과가 루프를
83
+ // 계속 돌면 턴당 violation 기록이 2-3배로 불어나 lifecycle T2 트리거
84
+ // (violations_30d>=3)를 조기 발화시키고 meta 승격을 영구 차단한다.
85
+ // 기록 카디널리티는 block 모드와 정확히 동일하게 보존한다.
76
86
  if (c.kind === 'block')
77
- break; // 첫 block 에서 중단 (이후 가드는 기록되지 않음)
87
+ break;
78
88
  }
79
89
  return results;
80
90
  }
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Forgen v0.5.0 — per-model 가드 프로필 (ADR-010 W4-3, F3)
3
+ *
4
+ * 근거 (v0.4.11 실측, docs/release/v0.4.11-calibration-pending.md):
5
+ * opus-4.8 에서 완료 가드(TEST-1/2/3) blocks=0 — easy N=10 / hard N=6,
6
+ * false-completion 압박 케이스 포함. 프론티어 모델은 스스로 정직해져서
7
+ * 완료 가드가 발화하지 않으며, 발화한다면 거짓양성일 개연성이 높다.
8
+ * → 측정된 모델은 block 대신 advise(기록+주입만)로 강등한다.
9
+ *
10
+ * 미측정 모델(sonnet-5 포함)은 보수적으로 block 유지 — R1/R2 재캘리브레이션이
11
+ * 측정을 제공하면 테이블을 갱신한다. DANGEROUS 가드(파괴 명령)는 모델 무관
12
+ * 결정적 안전장치라 이 프로필의 대상이 아니다.
13
+ *
14
+ * 모델 식별 경로 (probe 2026-07-16): hook stdin 에는 모델 필드가 없다.
15
+ * Claude Code statusline stdin 에는 session_id + model.id 가 오므로,
16
+ * `forgen statusline` 이 세션별 캐시를 남기고 Stop/SubagentStop 가드가
17
+ * session_id 로 조회한다. 캐시 부재(statusline 미사용 등) 시 'unknown'
18
+ * → 현행 동작(block) 유지. FORGEN_MODEL env 가 있으면 최우선.
19
+ */
20
+ export type CompletionGuardMode = 'block' | 'advise';
21
+ export declare function guardModeForModel(modelId: string | null | undefined): CompletionGuardMode;
22
+ /** statusline 이 세션별 모델을 기록 (fail-open) */
23
+ export declare function cacheSessionModel(sessionId: string, modelId: string, home?: string): void;
24
+ /** 가드가 세션별 모델 조회. 우선순위: FORGEN_MODEL env > statusline 캐시 > null */
25
+ export declare function readSessionModel(sessionId: string, home?: string): string | null;
@@ -0,0 +1,63 @@
1
+ /**
2
+ * Forgen v0.5.0 — per-model 가드 프로필 (ADR-010 W4-3, F3)
3
+ *
4
+ * 근거 (v0.4.11 실측, docs/release/v0.4.11-calibration-pending.md):
5
+ * opus-4.8 에서 완료 가드(TEST-1/2/3) blocks=0 — easy N=10 / hard N=6,
6
+ * false-completion 압박 케이스 포함. 프론티어 모델은 스스로 정직해져서
7
+ * 완료 가드가 발화하지 않으며, 발화한다면 거짓양성일 개연성이 높다.
8
+ * → 측정된 모델은 block 대신 advise(기록+주입만)로 강등한다.
9
+ *
10
+ * 미측정 모델(sonnet-5 포함)은 보수적으로 block 유지 — R1/R2 재캘리브레이션이
11
+ * 측정을 제공하면 테이블을 갱신한다. DANGEROUS 가드(파괴 명령)는 모델 무관
12
+ * 결정적 안전장치라 이 프로필의 대상이 아니다.
13
+ *
14
+ * 모델 식별 경로 (probe 2026-07-16): hook stdin 에는 모델 필드가 없다.
15
+ * Claude Code statusline stdin 에는 session_id + model.id 가 오므로,
16
+ * `forgen statusline` 이 세션별 캐시를 남기고 Stop/SubagentStop 가드가
17
+ * session_id 로 조회한다. 캐시 부재(statusline 미사용 등) 시 'unknown'
18
+ * → 현행 동작(block) 유지. FORGEN_MODEL env 가 있으면 최우선.
19
+ */
20
+ import * as fs from 'node:fs';
21
+ import * as os from 'node:os';
22
+ import * as path from 'node:path';
23
+ /**
24
+ * 측정 기반 기본 테이블. 버전 경계 안전 매칭 (리뷰 SEV-2: 단순 prefix 는
25
+ * 가상의 'claude-opus-4-80' 같은 미측정 후속 모델까지 매치한다) —
26
+ * 정확히 해당 버전이거나 뒤에 비숫자 구분자([·- 등)가 와야 한다.
27
+ * opus-4-8 만 측정됨(v0.4.11 blocks=0, easy+hard) — 그 외 전부 보수적 block.
28
+ */
29
+ const MEASURED_ADVISE_RES = Object.freeze([
30
+ /^claude-opus-4-8(?![0-9])/, // claude-opus-4-8, claude-opus-4-8[1m] — 4-80 은 불일치
31
+ ]);
32
+ export function guardModeForModel(modelId) {
33
+ if (!modelId)
34
+ return 'block'; // unknown → 현행 유지
35
+ return MEASURED_ADVISE_RES.some(re => re.test(modelId)) ? 'advise' : 'block';
36
+ }
37
+ function cachePath(sessionId, home) {
38
+ // sessionId 는 호출측에서 sanitize 된 값이어야 함 (경로 주입 방지)
39
+ return path.join(home, '.forgen', 'state', `current-model-${sessionId}.json`);
40
+ }
41
+ /** statusline 이 세션별 모델을 기록 (fail-open) */
42
+ export function cacheSessionModel(sessionId, modelId, home = os.homedir()) {
43
+ try {
44
+ const p = cachePath(sessionId, home);
45
+ fs.mkdirSync(path.dirname(p), { recursive: true });
46
+ fs.writeFileSync(p, `${JSON.stringify({ modelId, at: new Date().toISOString() })}\n`);
47
+ }
48
+ catch { /* fail-open */ }
49
+ }
50
+ /** 가드가 세션별 모델 조회. 우선순위: FORGEN_MODEL env > statusline 캐시 > null */
51
+ export function readSessionModel(sessionId, home = os.homedir()) {
52
+ const envModel = process.env.FORGEN_MODEL;
53
+ if (envModel && envModel.trim().length > 0)
54
+ return envModel.trim();
55
+ try {
56
+ const raw = fs.readFileSync(cachePath(sessionId, home), 'utf-8');
57
+ const parsed = JSON.parse(raw);
58
+ return typeof parsed.modelId === 'string' ? parsed.modelId : null;
59
+ }
60
+ catch {
61
+ return null;
62
+ }
63
+ }