@uzysjung/agent-harness 26.149.0 → 26.150.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,6 +2,7 @@
2
2
  import {
3
3
  CATEGORIES,
4
4
  CLI_BASES,
5
+ DEFAULT_OPTIONS,
5
6
  DEV_METHOD_SKILL_IDS,
6
7
  EXTERNAL_ASSETS,
7
8
  INTERNAL_BUNDLED_SKILL_IDS,
@@ -9,13 +10,14 @@ import {
9
10
  TRUST_TIER,
10
11
  assetCliSupport,
11
12
  assetCostRows,
13
+ buildAssetSpec,
12
14
  buildManifest,
13
15
  estimateTokens,
14
16
  formatResidentCostBlock,
15
17
  init_esm_shims,
16
18
  residentCost,
17
19
  resolveBundleRoot
18
- } from "./chunk-YSW3OLH4.js";
20
+ } from "./chunk-NKBUDHPC.js";
19
21
 
20
22
  // src/trust-tier-drift.ts
21
23
  init_esm_shims();
@@ -66,6 +68,7 @@ function classifyDrift(tier, stars) {
66
68
  export {
67
69
  CATEGORIES,
68
70
  CLI_BASES,
71
+ DEFAULT_OPTIONS,
69
72
  DEV_METHOD_SKILL_IDS,
70
73
  EXTERNAL_ASSETS,
71
74
  INTERNAL_BUNDLED_SKILL_IDS,
@@ -74,6 +77,7 @@ export {
74
77
  TRUST_TIER,
75
78
  assetCliSupport,
76
79
  assetCostRows,
80
+ buildAssetSpec,
77
81
  buildManifest,
78
82
  classifyDrift,
79
83
  driftTargets,
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/trust-tier-drift.ts"],"sourcesContent":["/**\n * A1 — Trust Tier star-drift 검출 데이터 + 순수 로직.\n *\n * TRUST_TIER 의 star 기반 라벨(vetted ≥ 1000★ / experimental < 1000★)이 실제 GitHub\n * star 와 어긋났는지(drift) 판정한다. `official` 은 star 무관(Anthropic 공식·하네스 자체)\n * 이라 검사 제외.\n *\n * repo 출처 = 각 자산 method (in-code authoritative — 주석이 아니라 실제 설치 source):\n * skill → method.source (\"owner/repo\" 또는 github URL)\n * plugin → method.marketplace (\"owner/repo\")\n * npm → NPM_REPO_OVERRIDE[id] (pkg 는 npm 명이므로 GitHub repo 를 별도 명시)\n *\n * fetch/네트워크는 본 모듈에 없음 — 순수 로직만(테스트 가능). 실 fetch 는\n * `scripts/trust-tier-drift.mjs` 가 담당.\n */\nimport { EXTERNAL_ASSETS, type ExternalAsset, TRUST_TIER } from \"./external-assets.js\";\n\n// v26.79.0 — gen-compatibility 의 카테고리 exhaustiveness 가드용 SSOT (하드코딩 drift 차단).\nexport { CATEGORIES } from \"./categories.js\";\n// v26.116.0 (ADR-043) — context-cost-report.mjs 가 dist 에서 비용 계측기를 읽도록 re-export.\n// v26.140.0 — formatResidentCostBlock: 리포트가 표를 직접 조립하지 않도록 표시 계약도 함께 노출.\nexport {\n assetCostRows,\n estimateTokens,\n formatResidentCostBlock,\n residentCost,\n resolveBundleRoot,\n} from \"./context-cost.js\";\n// v26.76.0 — gen-compatibility.mjs 가 dist 에서 자산 카탈로그+tier 를 읽도록 re-export.\n// v26.93.0 — DEV_METHOD_SKILL_IDS 추가: gen-compatibility 의 CLI scope override 를\n// 하드코딩 id 목록 대신 SSOT 에서 derive (no-false-ship drift 구조 차단).\n// v26.95.0 — INTERNAL_BUNDLED_SKILL_IDS (dev-method + opt-in gemini-consult) 로 CLI scope derive.\nexport {\n // v26.102.0 (ADR-031) — gen-compatibility 의 CLI 열이 도달 범위를 derive 하도록 re-export.\n assetCliSupport,\n DEV_METHOD_SKILL_IDS,\n EXTERNAL_ASSETS,\n INTERNAL_BUNDLED_SKILL_IDS,\n TRUST_TIER,\n} from \"./external-assets.js\";\nexport { buildManifest } from \"./manifest.js\";\n// v26.102.0 (ADR-031) — 도달 라벨(\"N-CLI\")의 N 을 derive 하기 위한 re-export (매직 넘버 금지).\nexport { CLI_BASES, TRACKS } from \"./types.js\";\n\n/** vetted 경계 (NORTH_STAR / PRD v26-71 D2). */\nexport const STAR_THRESHOLD = 1000;\n\nexport type StarTier = \"vetted\" | \"experimental\";\nexport type DriftVerdict = \"ok\" | \"promote\" | \"demote\";\n\n/**\n * method 가 GitHub repo 를 안 담는 자산(npm.pkg / npx-run.cmd 는 npm 명) → 트러스트 근거가\n * 된 GitHub repo 를 명시 매핑. override 가 method 도출보다 우선.\n */\nconst REPO_OVERRIDE: Record<string, string> = {\n \"vercel-cli\": \"vercel/vercel\", // npm\n \"netlify-cli\": \"netlify/cli\", // npm\n \"supabase-cli\": \"supabase/cli\", // npm\n \"agent-browser\": \"vercel-labs/agent-browser\", // npm\n openspec: \"Fission-AI/OpenSpec\", // npm (v26.75.0)\n \"bmad-method\": \"bmad-code-org/BMAD-METHOD\", // npx-run (v26.75.0)\n};\n\n/** \"https://github.com/owner/repo\" 또는 \"owner/repo[/...]\" → \"owner/repo\". 실패 시 null. */\nexport function normalizeRepo(source: string): string | null {\n const stripped = source.replace(/^https?:\\/\\/github\\.com\\//i, \"\");\n const m = stripped.match(/^([^/\\s]+\\/[^/\\s]+)/);\n return m?.[1] ?? null;\n}\n\n/** 자산의 GitHub owner/repo 도출. override 우선 → skill/plugin method. 도출 불가 시 null. */\nexport function repoForAsset(asset: ExternalAsset): string | null {\n const override = REPO_OVERRIDE[asset.id];\n if (override) return override;\n const m = asset.method;\n if (m.kind === \"skill\") return normalizeRepo(m.source);\n if (m.kind === \"plugin\") return normalizeRepo(m.marketplace);\n return null;\n}\n\nexport interface DriftTarget {\n id: string;\n tier: StarTier;\n repo: string;\n}\n\n/** star 기반(vetted/experimental) 자산만 + repo 도출 가능한 것만 검사 대상. */\nexport function driftTargets(\n assets: ReadonlyArray<ExternalAsset> = EXTERNAL_ASSETS,\n): DriftTarget[] {\n const out: DriftTarget[] = [];\n for (const a of assets) {\n const tier = TRUST_TIER[a.id];\n if (tier !== \"vetted\" && tier !== \"experimental\") continue;\n const repo = repoForAsset(a);\n if (!repo) continue; // 도출 불가 — 테스트가 0건을 강제하므로 정상 경로에선 발생 안 함\n out.push({ id: a.id, tier, repo });\n }\n return out;\n}\n\n/** 정적 tier 가 실제 star 와 어긋났는지 판정. */\nexport function classifyDrift(tier: StarTier, stars: number): DriftVerdict {\n if (tier === \"vetted\" && stars < STAR_THRESHOLD) return \"demote\";\n if (tier === \"experimental\" && stars >= STAR_THRESHOLD) return \"promote\";\n return \"ok\";\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;AAAA;AA6CO,IAAM,iBAAiB;AAS9B,IAAM,gBAAwC;AAAA,EAC5C,cAAc;AAAA;AAAA,EACd,eAAe;AAAA;AAAA,EACf,gBAAgB;AAAA;AAAA,EAChB,iBAAiB;AAAA;AAAA,EACjB,UAAU;AAAA;AAAA,EACV,eAAe;AAAA;AACjB;AAGO,SAAS,cAAc,QAA+B;AAC3D,QAAM,WAAW,OAAO,QAAQ,8BAA8B,EAAE;AAChE,QAAM,IAAI,SAAS,MAAM,qBAAqB;AAC9C,SAAO,IAAI,CAAC,KAAK;AACnB;AAGO,SAAS,aAAa,OAAqC;AAChE,QAAM,WAAW,cAAc,MAAM,EAAE;AACvC,MAAI,SAAU,QAAO;AACrB,QAAM,IAAI,MAAM;AAChB,MAAI,EAAE,SAAS,QAAS,QAAO,cAAc,EAAE,MAAM;AACrD,MAAI,EAAE,SAAS,SAAU,QAAO,cAAc,EAAE,WAAW;AAC3D,SAAO;AACT;AASO,SAAS,aACd,SAAuC,iBACxB;AACf,QAAM,MAAqB,CAAC;AAC5B,aAAW,KAAK,QAAQ;AACtB,UAAM,OAAO,WAAW,EAAE,EAAE;AAC5B,QAAI,SAAS,YAAY,SAAS,eAAgB;AAClD,UAAM,OAAO,aAAa,CAAC;AAC3B,QAAI,CAAC,KAAM;AACX,QAAI,KAAK,EAAE,IAAI,EAAE,IAAI,MAAM,KAAK,CAAC;AAAA,EACnC;AACA,SAAO;AACT;AAGO,SAAS,cAAc,MAAgB,OAA6B;AACzE,MAAI,SAAS,YAAY,QAAQ,eAAgB,QAAO;AACxD,MAAI,SAAS,kBAAkB,SAAS,eAAgB,QAAO;AAC/D,SAAO;AACT;","names":[]}
1
+ {"version":3,"sources":["../src/trust-tier-drift.ts"],"sourcesContent":["/**\n * A1 — Trust Tier star-drift 검출 데이터 + 순수 로직.\n *\n * TRUST_TIER 의 star 기반 라벨(vetted ≥ 1000★ / experimental < 1000★)이 실제 GitHub\n * star 와 어긋났는지(drift) 판정한다. `official` 은 star 무관(Anthropic 공식·하네스 자체)\n * 이라 검사 제외.\n *\n * repo 출처 = 각 자산 method (in-code authoritative — 주석이 아니라 실제 설치 source):\n * skill → method.source (\"owner/repo\" 또는 github URL)\n * plugin → method.marketplace (\"owner/repo\")\n * npm → NPM_REPO_OVERRIDE[id] (pkg 는 npm 명이므로 GitHub repo 를 별도 명시)\n *\n * fetch/네트워크는 본 모듈에 없음 — 순수 로직만(테스트 가능). 실 fetch 는\n * `scripts/trust-tier-drift.mjs` 가 담당.\n */\nimport { EXTERNAL_ASSETS, type ExternalAsset, TRUST_TIER } from \"./external-assets.js\";\n\n// v26.79.0 — gen-compatibility 의 카테고리 exhaustiveness 가드용 SSOT (하드코딩 drift 차단).\nexport { CATEGORIES } from \"./categories.js\";\n// v26.116.0 (ADR-043) — context-cost-report.mjs 가 dist 에서 비용 계측기를 읽도록 re-export.\n// v26.140.0 — formatResidentCostBlock: 리포트가 표를 직접 조립하지 않도록 표시 계약도 함께 노출.\nexport {\n assetCostRows,\n estimateTokens,\n formatResidentCostBlock,\n residentCost,\n resolveBundleRoot,\n} from \"./context-cost.js\";\n// v26.76.0 — gen-compatibility.mjs 가 dist 에서 자산 카탈로그+tier 를 읽도록 re-export.\n// v26.93.0 — DEV_METHOD_SKILL_IDS 추가: gen-compatibility 의 CLI scope override 를\n// 하드코딩 id 목록 대신 SSOT 에서 derive (no-false-ship drift 구조 차단).\n// v26.95.0 — INTERNAL_BUNDLED_SKILL_IDS (dev-method + opt-in gemini-consult) 로 CLI scope derive.\nexport {\n // v26.102.0 (ADR-031) — gen-compatibility 의 CLI 열이 도달 범위를 derive 하도록 re-export.\n assetCliSupport,\n DEV_METHOD_SKILL_IDS,\n EXTERNAL_ASSETS,\n INTERNAL_BUNDLED_SKILL_IDS,\n TRUST_TIER,\n} from \"./external-assets.js\";\nexport { buildAssetSpec, buildManifest } from \"./manifest.js\";\n// v26.102.0 (ADR-031) — 도달 라벨(\"N-CLI\")의 N 을 derive 하기 위한 re-export (매직 넘버 금지).\nexport { CLI_BASES, DEFAULT_OPTIONS, TRACKS } from \"./types.js\";\n\n/** vetted 경계 (NORTH_STAR / PRD v26-71 D2). */\nexport const STAR_THRESHOLD = 1000;\n\nexport type StarTier = \"vetted\" | \"experimental\";\nexport type DriftVerdict = \"ok\" | \"promote\" | \"demote\";\n\n/**\n * method 가 GitHub repo 를 안 담는 자산(npm.pkg / npx-run.cmd 는 npm 명) → 트러스트 근거가\n * 된 GitHub repo 를 명시 매핑. override 가 method 도출보다 우선.\n */\nconst REPO_OVERRIDE: Record<string, string> = {\n \"vercel-cli\": \"vercel/vercel\", // npm\n \"netlify-cli\": \"netlify/cli\", // npm\n \"supabase-cli\": \"supabase/cli\", // npm\n \"agent-browser\": \"vercel-labs/agent-browser\", // npm\n openspec: \"Fission-AI/OpenSpec\", // npm (v26.75.0)\n \"bmad-method\": \"bmad-code-org/BMAD-METHOD\", // npx-run (v26.75.0)\n};\n\n/** \"https://github.com/owner/repo\" 또는 \"owner/repo[/...]\" → \"owner/repo\". 실패 시 null. */\nexport function normalizeRepo(source: string): string | null {\n const stripped = source.replace(/^https?:\\/\\/github\\.com\\//i, \"\");\n const m = stripped.match(/^([^/\\s]+\\/[^/\\s]+)/);\n return m?.[1] ?? null;\n}\n\n/** 자산의 GitHub owner/repo 도출. override 우선 → skill/plugin method. 도출 불가 시 null. */\nexport function repoForAsset(asset: ExternalAsset): string | null {\n const override = REPO_OVERRIDE[asset.id];\n if (override) return override;\n const m = asset.method;\n if (m.kind === \"skill\") return normalizeRepo(m.source);\n if (m.kind === \"plugin\") return normalizeRepo(m.marketplace);\n return null;\n}\n\nexport interface DriftTarget {\n id: string;\n tier: StarTier;\n repo: string;\n}\n\n/** star 기반(vetted/experimental) 자산만 + repo 도출 가능한 것만 검사 대상. */\nexport function driftTargets(\n assets: ReadonlyArray<ExternalAsset> = EXTERNAL_ASSETS,\n): DriftTarget[] {\n const out: DriftTarget[] = [];\n for (const a of assets) {\n const tier = TRUST_TIER[a.id];\n if (tier !== \"vetted\" && tier !== \"experimental\") continue;\n const repo = repoForAsset(a);\n if (!repo) continue; // 도출 불가 — 테스트가 0건을 강제하므로 정상 경로에선 발생 안 함\n out.push({ id: a.id, tier, repo });\n }\n return out;\n}\n\n/** 정적 tier 가 실제 star 와 어긋났는지 판정. */\nexport function classifyDrift(tier: StarTier, stars: number): DriftVerdict {\n if (tier === \"vetted\" && stars < STAR_THRESHOLD) return \"demote\";\n if (tier === \"experimental\" && stars >= STAR_THRESHOLD) return \"promote\";\n return \"ok\";\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAAA;AA6CO,IAAM,iBAAiB;AAS9B,IAAM,gBAAwC;AAAA,EAC5C,cAAc;AAAA;AAAA,EACd,eAAe;AAAA;AAAA,EACf,gBAAgB;AAAA;AAAA,EAChB,iBAAiB;AAAA;AAAA,EACjB,UAAU;AAAA;AAAA,EACV,eAAe;AAAA;AACjB;AAGO,SAAS,cAAc,QAA+B;AAC3D,QAAM,WAAW,OAAO,QAAQ,8BAA8B,EAAE;AAChE,QAAM,IAAI,SAAS,MAAM,qBAAqB;AAC9C,SAAO,IAAI,CAAC,KAAK;AACnB;AAGO,SAAS,aAAa,OAAqC;AAChE,QAAM,WAAW,cAAc,MAAM,EAAE;AACvC,MAAI,SAAU,QAAO;AACrB,QAAM,IAAI,MAAM;AAChB,MAAI,EAAE,SAAS,QAAS,QAAO,cAAc,EAAE,MAAM;AACrD,MAAI,EAAE,SAAS,SAAU,QAAO,cAAc,EAAE,WAAW;AAC3D,SAAO;AACT;AASO,SAAS,aACd,SAAuC,iBACxB;AACf,QAAM,MAAqB,CAAC;AAC5B,aAAW,KAAK,QAAQ;AACtB,UAAM,OAAO,WAAW,EAAE,EAAE;AAC5B,QAAI,SAAS,YAAY,SAAS,eAAgB;AAClD,UAAM,OAAO,aAAa,CAAC;AAC3B,QAAI,CAAC,KAAM;AACX,QAAI,KAAK,EAAE,IAAI,EAAE,IAAI,MAAM,KAAK,CAAC;AAAA,EACnC;AACA,SAAO;AACT;AAGO,SAAS,cAAc,MAAgB,OAA6B;AACzE,MAAI,SAAS,YAAY,QAAQ,eAAgB,QAAO;AACxD,MAAI,SAAS,kBAAkB,SAAS,eAAgB,QAAO;AAC/D,SAAO;AACT;","names":[]}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uzysjung/agent-harness",
3
- "version": "26.149.0",
3
+ "version": "26.150.0",
4
4
  "description": "Curate vetted AI-coding skills & plugins by your tech stack — install only what you need, across Claude Code, Codex, OpenCode & Antigravity",
5
5
  "type": "module",
6
6
  "publishConfig": {
@@ -27,9 +27,63 @@ fi
27
27
  # 실패 모드이기 때문이다(cli-development.md §Cross-Platform).
28
28
  ORPHAN_NOTE=""
29
29
  PROJ_DIR=$(pwd)
30
- ORPHANS=$(ps -eo pid,ppid,command 2>/dev/null | awk -v d="$PROJ_DIR" '$2==1 && index($0,d)>0 {n++} END{print n+0}')
31
- if [ "${ORPHANS:-0}" -gt 0 ]; then
32
- ORPHAN_NOTE=" WARNING: ${ORPHANS} orphaned process(es) from a previous session still reference this project (parent died, reparented to init). Inspect with: ps -eo pid,ppid,etime,command | grep \$(pwd) | grep -v grep — then stop what you recognise. Leaving them costs memory and can hold ports or file locks."
30
+ # **물리 경로도 함께 본다.** macOS 의 `/var` 는 `/private/var` 로의 심링크라 `pwd` 는
31
+ # `/var/...`, 커널·`lsof` 는 `/private/var/...` 를 낸다. 논리 경로만 대면 접두사가 어긋나
32
+ # 실재하는 고아를 **0건으로** 보고한다 — 실제로 그렇게 짰다가 진짜 고아를 만든 시험에서 잡혔다.
33
+ PROJ_DIR_P=$(pwd -P 2>/dev/null || printf '%s' "$PROJ_DIR")
34
+
35
+ # **한 번만 찍고 그 스냅샷을 읽는다.** `ps | grep <이름>` 은 grep 자신의 커맨드라인을 매치해
36
+ # 항상 양성을 낸다 — 이 리포가 실제로 오판한 형태다.
37
+ PS_SNAP=$(ps -eo pid,ppid,command 2>/dev/null || true)
38
+
39
+ # ⓐ 커맨드라인에 프로젝트 경로가 **든** 고아 (`npm run dev /path/...` 처럼 경로를 인자로 받은 것).
40
+ ORPHANS=$(printf '%s\n' "$PS_SNAP" | awk -v d="$PROJ_DIR" '$2==1 && index($0,d)>0 {n++} END{print n+0}')
41
+
42
+ # ⓑ 커맨드라인에 경로가 **없는** 고아. 서브에이전트가 이 모양이다 — 경로는 cwd 에만 있어서
43
+ # ⓐ 는 이 부류를 **구조적으로 0건**으로 보고했다(#326 실측: 살아 있는 서브에이전트 2건 → 0건).
44
+ # 0 을 내는 탐지기는 없는 것보다 나쁘다. 거짓 안심을 준다.
45
+ #
46
+ # **비용 때문에 후보를 먼저 좁힌다.** cwd 를 묻는 것은 macOS 에서 pid 당 약 4 ms 다(실측).
47
+ # ppid=1 전체(이 머신 433개)에 물으면 세션 시작이 배로 느려지고, 그 비용은 설치받은
48
+ # 사람이 **매 세션** 낸다. 그래서 ⓐ 가 이미 세지 않은 것 중 **에이전트 부류만** 보고,
49
+ # 후보가 0이면 cwd 를 아예 안 묻는다. 상한 40개.
50
+ #
51
+ # **비용의 원인을 한 번 잘못 짚었다.** 처음엔 후보 수 탓인 줄 알고 범위를 좁혔는데, 재보니
52
+ # **후보 1개짜리 `lsof` 가 1,444 ms** 였다 — 비싼 것은 후보 수가 아니라 `lsof` 자신이었다.
53
+ # `-b`(블록 가능 호출 회피)를 붙이자 같은 답에 **74 ms** 가 됐고, 그래서 범위를 다시 넓힐 수
54
+ # 있었다. 좁힌 채로 뒀으면 경로가 argv 에 없는 비-에이전트 고아(디렉터리 안에서 띄운
55
+ # `node server.js` 등)를 영영 못 봤을 것이다.
56
+ CAND=$(printf '%s\n' "$PS_SNAP" | awk -v d="$PROJ_DIR" \
57
+ '$2==1 && index($0,d)==0 && /claude|node|python|bun|deno|--agent-name/ {print $1}' | head -40)
58
+ CWD_ORPHANS=0
59
+ if [ -n "$CAND" ]; then
60
+ if [ -d /proc ]; then
61
+ # Linux: /proc 는 사실상 공짜다.
62
+ for pid in $CAND; do
63
+ cwd=$(readlink "/proc/$pid/cwd" 2>/dev/null || true)
64
+ case "$cwd" in
65
+ "$PROJ_DIR" | "$PROJ_DIR"/* | "$PROJ_DIR_P" | "$PROJ_DIR_P"/*) CWD_ORPHANS=$((CWD_ORPHANS + 1)) ;;
66
+ esac
67
+ done
68
+ elif command -v lsof > /dev/null 2>&1; then
69
+ # macOS: /proc 이 없다. 한 번에 묻고, 접두사 일치로 **이 프로젝트 밑**만 센다 —
70
+ # 다른 프로젝트의 고아를 남의 세션이 판단해선 안 된다.
71
+ # `-b` 가 이 검사를 쓸 수 있게 만든다 — 블록 가능한 커널 호출을 피한다. 실측: 후보 1개에
72
+ # `-b` 없이 **1,444 ms**, 있으면 **74 ms**(19배)이고 답은 같다(cwd 28건 동일). `-w` 는 그때
73
+ # 나는 경고를 죽이고, `-n`·`-P` 는 우리가 안 쓰는 이름 해석을 건너뛴다.
74
+ CWD_ORPHANS=$(lsof -b -w -n -P -p "$(printf '%s' "$CAND" | tr '\n' ',' | sed 's/,$//')" -a -d cwd -Fn 2>/dev/null \
75
+ | awk -v d="$PROJ_DIR" -v dp="$PROJ_DIR_P" '
76
+ substr($0,1,1)=="n" {
77
+ p = substr($0,2)
78
+ if (p==d || index(p, d "/")==1 || p==dp || index(p, dp "/")==1) n++
79
+ } END {print n+0}')
80
+ fi
81
+ # lsof 도 /proc 도 없으면 CWD_ORPHANS 는 0 이다. 이 경우는 **못 본 것**이지 없는 것이 아니다.
82
+ fi
83
+
84
+ ORPHAN_TOTAL=$((${ORPHANS:-0} + ${CWD_ORPHANS:-0}))
85
+ if [ "$ORPHAN_TOTAL" -gt 0 ]; then
86
+ ORPHAN_NOTE=" WARNING: ${ORPHAN_TOTAL} orphaned process(es) from a previous session still belong to this project (parent died, reparented to init); ${CWD_ORPHANS} of them are only visible by working directory, which is how subagents look. Inspect with: ps -eo pid,ppid,etime,command | grep \$(pwd) | grep -v grep — and for the rest, resolve each candidate's cwd. Then stop what you recognise. Leaving them costs memory and can hold ports or file locks."
33
87
  fi
34
88
 
35
89
  # 3. 세션 컨텍스트 출력
@@ -7,13 +7,12 @@ description: >-
7
7
  user-perspective (UX) — and enumerates concrete, severity-ranked gaps; BENCHMARK verifies how a
8
8
  reference service closes each one and PROPOSES a differentiated close. VERIFY, CHANGE-IMPACT,
9
9
  DRIFT and FULL extend the same loop to post-fix closure, baseline changes, doc-vs-code drift, and
10
- the whole-service sweep. Use when the user says any of: "북극성 기준으로 부족한 점", "사용자
11
- 관점에서 부족한 점", "다른 벤치마크 서비스는 이 부분을 어떻게 해결했는지", "갭분석", "레퍼런스
12
- 서비스랑 비교해서 부족한 점 찾아줘" — or the English equivalents: "gap analysis", "what are we
13
- missing vs the ideal/north-star", "benchmark against reference services", "audit this service".
14
- Fires for both Korean and English phrasing. Do NOT use it to *define* product direction (that is
15
- north-star), to review ONE standalone artifact's prose (that is multi-persona-review), or to turn
16
- an unverified benchmark claim into a fact.
10
+ the whole-service sweep. Use when the user says any of: "북극성 기준으로 부족한 점", "갭분석",
11
+ "다른 벤치마크 서비스는 이 부분을 어떻게 해결했는지", "레퍼런스 서비스랑 비교해서 부족한 점
12
+ 찾아줘" — or the English "gap analysis", "benchmark against reference services", "audit this
13
+ service". Do NOT use it to *define* product direction (that is north-star), to review ONE
14
+ standalone artifact's prose (that is multi-persona-review), or to turn an unverified benchmark
15
+ claim into a fact.
17
16
  ---
18
17
 
19
18
  # Audit Service Gaps (reverse + competitive)
@@ -4,21 +4,16 @@ description: >-
4
4
  Make a technical explanation land, and turn the decision at the end of it into
5
5
  something the reader can approve in one pass. Two halves of one job: (1) EXPLAIN
6
6
  — fix the referent first (one name often points at two things), lead with who is
7
- affected and what changes, put file paths and symbols after the claim as evidence;
7
+ affected and what changes, put file paths after the claim as evidence;
8
8
  (2) DECIDE — present approval/choice moments in the user's four-part format
9
- 전후맥락 (context) → 추천 + 이유 (recommendation) → UI/UX 형태 (a scannable
10
- table/option-list) → ASIS→TOBE contrast, led by the recommendation so the user can
9
+ 전후맥락 (context) → 추천 + 이유 (recommendation) → UI/UX 형태 (a scannable table) → ASIS→TOBE contrast, led by the recommendation so the user can
11
10
  say yes fast. Run it whenever you explain a bug, a cause, or what your change did
12
- — especially the moment the reader says they don't follow ("뭔 소리야", "쉽게
13
- 설명해줘", "이해가 안 돼", "I don't follow", "in plain terms"), or when your draft
14
- opens with a file path or symbol name — and whenever you are about to ask
15
- "should I do this?". Triggers on the user's verbatim phrases "ASIS TOBE로 설명",
16
- "ASIS-TOBE로 알려줘", "화면으로 ASIS TOBE로 설명", "의사결정 / 컨펌 요청",
17
- "이거 진행할까요?", the softer "다음 진행할 것들 알려줘", and the English
18
- equivalents "present this as ASIS/TOBE", "give me the as-is to-be", "should I do
19
- A or B", "ask for my approval", "lay out the options". Do NOT fire for pure
20
- information with no decision in it, for trivial reversible actions you would just
21
- do, or for context-free word/sentence translation — that is ordinary translation.
11
+ — especially the moment the reader says they don't follow ("뭔 소리야", "쉽게 설명해줘",
12
+ "이해가 안 돼", "I don't follow") — and whenever you are about to ask "should I do
13
+ this?" ("ASIS TOBE로 설명", "이거 진행할까요?", "should I do A or B"). Do NOT fire
14
+ for pure information with no decision in it, for trivial reversible actions you
15
+ would just do, or for context-free word/sentence translation — that is ordinary
16
+ translation.
22
17
  ---
23
18
 
24
19
  # Clear Korean Communication
@@ -55,21 +55,38 @@ Do not run after compaction without first reconstructing the best available stat
55
55
  - Historical archives are disabled by default. If explicitly required, keep them under
56
56
  `.handoff/archive/`, apply a retention limit, and never auto-load them on resume.
57
57
 
58
- ### `MEMORY.md` — durable facts only
59
-
60
- Keep:
61
-
62
- - stable purpose, scope, constraints, invariants, and operating principles;
63
- - durable repository or user preferences;
64
- - pointers to active ADRs and `.handoff/CURRENT.md`.
65
-
66
- Remove or exclude:
67
-
68
- - current branch, test failure, task progress, temporary blocker, raw output;
69
- - completed session history, old anchors, duplicated repository content.
70
-
71
- Update or replace existing entries; do not append near-duplicates. Remove superseded facts. Target:
72
- **200 lines / 20 KB maximum**, unless the repository defines another limit.
58
+ ### `MEMORY.md` — rules, not history
59
+
60
+ Memory is loaded **every session**, so every line is a standing cost. What earns that cost is
61
+ **principles, recurrence countermeasures, and facts you actually need to do the work** — not a
62
+ record of what was done.
63
+
64
+ **Judge every entry — the ones you are adding AND the ones already there — with three questions:**
65
+
66
+ 1. **Does it change what I do next time?** If not, don't write it. "We shipped X" changes nothing.
67
+ 2. **Does it already live somewhere?** Rules, the project's instruction files, ADRs, skills, and
68
+ git history are each a source of truth. If the fact is there, that place owns it — do not keep
69
+ a copy here. **A duplicated fact is guaranteed to rot on one side**, and you cannot tell which.
70
+ 3. **Is it finished?** Completed cycles, release logs, and version history belong to the
71
+ changelog, ADRs, and git — not here.
72
+
73
+ What survives all three: **operating principles · countermeasures for repeated mistakes ·
74
+ facts that cannot be derived from the repository** (another tool's flags, limits, and policies;
75
+ standing decisions such as "we accepted this risk, do not re-open it").
76
+
77
+ **Re-judge the whole index at every handoff, not just the new lines.** An index only ever grows
78
+ unless something forces the question, and this is that moment.
79
+
80
+ Prefer updating an existing entry over adding a near-duplicate. To drop one, **move the file to
81
+ `archive/` rather than deleting it** — it leaves the index (so it stops loading) while staying
82
+ recoverable.
83
+
84
+ **Size is a symptom, not the standard.** Keep the index under **200 lines / 20 KB** (unless the
85
+ repository sets another limit), but being under it is *not* evidence the index is healthy: a short
86
+ index full of duplicates and finished history still fails all three questions. Measured case: an
87
+ index at 67 lines / 20.7 KB passed the size rule while **48 of its entries were dead** — completed
88
+ cycle records and copies of facts already owned by rules and ADRs. Re-judged against the three
89
+ questions, it came out at 23 lines / 6.0 KB.
73
90
 
74
91
  ### `docs/decisions/` — durable decisions only
75
92
 
@@ -1,30 +1,19 @@
1
1
  ---
2
2
  name: external-model-consult
3
3
  description: >-
4
- Consult a second, non-Claude model through a bundled wrapper for the four things
5
- an external round-trip actually buys: (1) natural, native-sounding KOREAN phrasing
6
- via Google Gemini (Antigravity `agy` CLI) — copy, UI microcopy, marketing/brochure
7
- text, toasts, user-facing messages, translations, rewrites; (2) a MULTI-PERSONA /
8
- second-opinion review of a design, plan, spec, PR, or piece of writing; (3) CONCISE,
9
- well-STRUCTURED writing via OpenAI Codex (`codex exec`) — tightening verbose prose,
10
- restructuring a doc into a clean outline / tables / sections, executive summaries,
11
- README skeletons, changelog entries; and (4) IMAGE GENERATION as real PNG/JPG files
12
- on disk (labeled flowcharts / architecture / sequence diagrams are NOT this — render
13
- those natively as Mermaid). Use whenever Korean text needs to read naturally rather
14
- than translated, whenever the user says the Korean "sounds awkward / 어색해 /
15
- 자연스럽게 다듬어줘", whenever you are about to hand-write polished Korean copy
16
- yourself, whenever a document needs to get SHORTER and better ORGANIZED (not
17
- prettier-sounding), whenever you want an independent non-Claude critique, or when
18
- the user asks for a generated image. Triggers on "gemini 한테 물어봐 / gemini 로
19
- 다듬어 / gemini로 이미지 만들어줘 / nano banana / agy / antigravity / 제3자 관점 /
20
- second opinion / codex한테 물어봐 / codex로 정리해 / 간결하게 정리해줘 / 구조화해줘 /
21
- 문서 구조 잡아줘 / 이미지 만들어줘 / 그림 생성해줘", and in English "ask codex",
22
- "ask gemini", "tighten this up", "make this concise", "restructure this doc",
23
- "generate an image". Returns candidates/files for the user to choose from; never
24
- auto-applies. Do NOT use for deterministic transforms (rename, reformat, sort),
25
- labeled diagrams, internal logs or code identifiers, anything needing repo secrets,
26
- a native parallel-subagent panel (that is `multi-persona-review` where installed),
27
- or when the user explicitly wants YOUR answer.
4
+ Consult a second, non-Claude model through a bundled wrapper for four things:
5
+ (1) natural, native-sounding KOREAN phrasing via Google Gemini — copy, UI microcopy,
6
+ marketing text; (2) a MULTI-PERSONA / second-opinion review of a design, plan or spec;
7
+ (3) CONCISE, well-STRUCTURED writing via OpenAI Codex — tightening prose, restructuring
8
+ a doc; and (4) IMAGE GENERATION as real image files on disk (not labeled diagrams —
9
+ use Mermaid). Fire it when Korean reads translated, or when you would otherwise
10
+ hand-write polished Korean yourself. Triggers on "어색해", "자연스럽게 다듬어줘",
11
+ "gemini 한테 물어봐", "nano banana", "제3자 관점", "codex한테 물어봐",
12
+ "간결하게 정리해줘", "이미지 만들어줘", and in English "ask gemini", "ask codex",
13
+ "generate an image". Returns candidates to choose from; never auto-applies. Do NOT use
14
+ for deterministic transforms (rename, reformat, sort), labeled diagrams, internal logs
15
+ or identifiers, anything needing repo secrets, a native subagent panel
16
+ (`multi-persona-review`), or when the user explicitly wants YOUR answer.
28
17
  ---
29
18
 
30
19
  # external-model-consult
@@ -2,21 +2,17 @@
2
2
  name: model-orchestration
3
3
  description: >-
4
4
  Apply the fixed model-role and thinking-effort policy whenever work is delegated to subagents
5
- or a model/effort choice is made: the orchestrator (top-tier model, Fable) DIRECTLY owns 설계·
6
- 기획·분배·리뷰 — sets service direction, arbitrates and reviews plan/spec documents (with
7
- multi-persona-review), improves shipped features, and hunts performance/security problems;
8
- core implementation, test authoring/execution (E2E included), and code verification/V&V go to
9
- Opus at xhigh or above; repetitive/simple implementation (applying an established pattern)
10
- goes to Sonnet at high or above — Sonnet is never used for tests or verification;
11
- plan/spec drafts may be produced by Opus from a Fable direction brief, but the
12
- review/decision is always Fable's. Never delegate below the effort floors. Use whenever you
13
- are about to spawn an Agent/Task/Workflow worker, pick a model for a subtask, set a
14
- thinking/effort level, assign verification, or hand off orchestration because the current
15
- model's quota is exhausted. Trigger on "위임해", "에이전트로 돌려", "오케스트레이션",
16
- "모델 역할분담", "어떤 모델로", "effort 얼마로", "thinking level", "서브에이전트", or in English
17
- "delegate this", "spawn an agent for", "which model should", "route this task", "verify with",
18
- "orchestrate". Fire even when the user doesn't name the policy — any delegation decision is
19
- in scope.
5
+ or a model/effort choice is made: the orchestrator (top-tier model, Fable) DIRECTLY owns
6
+ 설계·기획·분배·리뷰 — sets direction, reviews plan/spec docs, improves shipped features, and hunts
7
+ perf/security problems; core implementation, test authoring/execution
8
+ (E2E included), and code verification/V&V go to Opus at xhigh or above;
9
+ repetitive/simple implementation goes to Sonnet at high or above — Sonnet is never used for
10
+ tests or verification; plan/spec drafts may be produced by Opus from a Fable direction brief,
11
+ but the review/decision is always Fable's. Never delegate below the effort floors. Use whenever
12
+ you are about to spawn an Agent/Task/Workflow worker, pick a model for a subtask, set a
13
+ thinking/effort level, or assign verification. Trigger on "위임해", "오케스트레이션", "모델 역할분담", "어떤 모델로",
14
+ "effort 얼마로", "서브에이전트", "delegate this", "which model should". Fire even when the
15
+ user doesn't name the policy — any delegation is in scope.
20
16
  ---
21
17
 
22
18
  # Model Orchestration Policy
@@ -10,12 +10,10 @@ description: >-
10
10
  should go next. Sits one layer above SPEC/PRD — answers 'why and where to', not
11
11
  'what and how'. Fires on the user's real phrasings: "앞으로 어떤 방향으로
12
12
  개선·발전시킬지 고민해봐", "NORTH.md / NORTH_STAR 보고 나아갈 방향 + 기능 제안",
13
- "나아갈 방향 + 기능제안 (수용 → 계획 수립하고 메모리에 기록)", "북극성 정렬 로드맵",
14
- as well as the English equivalents "what direction should we take next", "propose
15
- a roadmap / feature backlog from the north star", "plan the next milestones and
16
- save it to memory". Do NOT use it to find what is broken right now — detecting
17
- bugs, gaps, or quality regressions belongs to the audit/gap skills; this skill
18
- consumes their findings and DIRECTS forward planning.
13
+ "북극성 정렬 로드맵", and the English "what direction should we take next",
14
+ "propose a roadmap from the north star". Do NOT use it to find what is broken
15
+ right now — detecting bugs, gaps, or quality regressions belongs to the audit/gap
16
+ skills; this skill consumes their findings and DIRECTS forward planning.
19
17
  ---
20
18
 
21
19
  # North Star
@@ -4,17 +4,15 @@ description: >-
4
4
  Rewrite a task request into the canonical XML brief — objective, inputs, invariants,
5
5
  success_criteria, boundaries, autonomy, verification, communication, output_format — so the
6
6
  worker receives one judgeable definition of done instead of prose. Runs in two directions:
7
- INBOUND reshapes a sprawling or half-formed request (pasted requirements, a wall of background,
8
- a request that grew across several messages) into that shape, filling each field from context
9
- already on screen and deleting sections that do not apply; OUTBOUND writes the prompt that a
10
- spawned worker actually receives, so nobody hand-rolls a one-off prompt shape per delegation.
11
- Trigger on "브리프로 정리", "작업 지시서로 만들어", "프롬프트 구조화", "브리프 만들어줘",
12
- "이 요청 정리해줘", and in English "turn this into a task brief", "structure this prompt",
13
- "write the brief for this", "draft the spawn prompt". Fire unprompted the moment you are about
14
- to hand a multi-part task to a subagent, a workflow worker, or a parallel lane. Do NOT fire on a
15
- one-line question, a lookup, or an ordinary conversational exchange where you simply need one
16
- more piece of information — asking a clarifying question is not a brief, and wrapping a
17
- one-sentence ask in nine XML tags costs more than it buys.
7
+ INBOUND reshapes a sprawling or half-formed request into that shape, filling each field from
8
+ context already on screen and deleting sections that do not apply; OUTBOUND writes the prompt
9
+ that a spawned worker actually receives. Trigger on "브리프로 정리", "작업 지시서로 만들어",
10
+ "프롬프트 구조화", and in English "turn this into a task brief", "structure this prompt".
11
+ Fire unprompted the moment you are about to hand a multi-part task to a subagent, a workflow
12
+ worker, or a parallel lane. Do NOT fire on a one-line question, a lookup, or an ordinary
13
+ conversational exchange where you simply need one more piece of information — asking a
14
+ clarifying question is not a brief, and wrapping a one-sentence ask in nine XML tags costs
15
+ more than it buys.
18
16
  ---
19
17
 
20
18
  # Task Brief