@uzysjung/agent-harness 26.149.0 → 26.151.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.ko.md +1 -1
- package/README.md +1 -1
- package/dist/{chunk-YSW3OLH4.js → chunk-3QBHZUVB.js} +164 -66
- package/dist/chunk-3QBHZUVB.js.map +1 -0
- package/dist/index.js +397 -293
- package/dist/index.js.map +1 -1
- package/dist/trust-tier-drift.js +5 -1
- package/dist/trust-tier-drift.js.map +1 -1
- package/package.json +1 -1
- package/templates/CLAUDE.md +145 -164
- package/templates/agents/build-error-resolver.md +1 -1
- package/templates/agents/plan-checker.md +1 -1
- package/templates/agents/reviewer.md +4 -5
- package/templates/antigravity/AGENTS.md.template +3 -23
- package/templates/codex/AGENTS.md.template +5 -56
- package/templates/hooks/protect-files.sh +4 -0
- package/templates/hooks/session-start.sh +57 -3
- package/templates/opencode/AGENTS.md.template +4 -52
- package/templates/opencode/opencode.json.template +0 -8
- package/templates/rules/change-management.md +0 -1
- package/templates/rules/cli-development.md +1 -1
- package/templates/rules/doc-governance.md +2 -0
- package/templates/rules/git-policy.md +1 -1
- package/templates/rules/ship-checklist.md +3 -3
- package/templates/rules/test-policy.md +3 -8
- package/templates/settings.json +1 -16
- package/templates/skills/agent-introspection-debugging/SKILL.md +1 -1
- package/templates/skills/audit-harness-fit/README.md +113 -0
- package/templates/skills/audit-harness-fit/SKILL.md +64 -433
- package/templates/skills/audit-harness-fit/evals/scenarios.yaml +222 -0
- package/templates/skills/audit-harness-fit/references/apply.md +66 -0
- package/templates/skills/audit-harness-fit/references/audit.md +160 -0
- package/templates/skills/audit-harness-fit/references/populate.md +74 -0
- package/templates/skills/audit-harness-fit/references/verification.md +123 -0
- package/templates/skills/audit-service-gaps/SKILL.md +6 -7
- package/templates/skills/clear-korean-communication/SKILL.md +8 -13
- package/templates/skills/compaction-handoff/SKILL.md +29 -12
- package/templates/skills/external-model-consult/SKILL.md +13 -24
- package/templates/skills/model-orchestration/SKILL.md +18 -15
- package/templates/skills/natural-korean/SKILL.md +45 -0
- package/templates/skills/north-star/SKILL.md +4 -6
- package/templates/skills/north-star/references/roadmap-method.md +2 -2
- package/templates/skills/{task-brief → objective-brief}/SKILL.md +17 -18
- package/templates/skills/recurrence-prevention/SKILL.md +16 -16
- package/dist/chunk-YSW3OLH4.js.map +0 -1
- package/templates/agents/code-reviewer.md +0 -237
- package/templates/agents/security-reviewer.md +0 -108
- package/templates/hooks/task-brief-nudge.sh +0 -57
- package/templates/skills/audit-harness-fit/references/official-criteria.md +0 -367
- package/templates/skills/continuous-learning-v2/SKILL.md +0 -361
- package/templates/skills/continuous-learning-v2/agents/observer-loop.sh +0 -362
- package/templates/skills/continuous-learning-v2/agents/observer.md +0 -189
- package/templates/skills/continuous-learning-v2/agents/session-guardian.sh +0 -150
- package/templates/skills/continuous-learning-v2/agents/start-observer.sh +0 -252
- package/templates/skills/continuous-learning-v2/config.json +0 -8
- package/templates/skills/continuous-learning-v2/hooks/observe.sh +0 -585
- package/templates/skills/continuous-learning-v2/scripts/detect-project.sh +0 -322
- package/templates/skills/continuous-learning-v2/scripts/instinct-cli.py +0 -1956
- package/templates/skills/continuous-learning-v2/scripts/lib/homunculus-dir.sh +0 -31
- package/templates/skills/continuous-learning-v2/scripts/migrate-homunculus.sh +0 -68
- package/templates/skills/continuous-learning-v2/scripts/test_parse_instinct.py +0 -1420
- package/templates/skills/humanize-korean/SKILL.md +0 -228
- package/templates/skills/spec-scaling/SKILL.md +0 -89
- package/templates/skills/strategic-compact/SKILL.md +0 -145
- package/templates/skills/strategic-compact/suggest-compact.sh +0 -54
package/dist/trust-tier-drift.js
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
import {
|
|
3
3
|
CATEGORIES,
|
|
4
4
|
CLI_BASES,
|
|
5
|
+
DEFAULT_OPTIONS,
|
|
5
6
|
DEV_METHOD_SKILL_IDS,
|
|
6
7
|
EXTERNAL_ASSETS,
|
|
7
8
|
INTERNAL_BUNDLED_SKILL_IDS,
|
|
@@ -9,13 +10,14 @@ import {
|
|
|
9
10
|
TRUST_TIER,
|
|
10
11
|
assetCliSupport,
|
|
11
12
|
assetCostRows,
|
|
13
|
+
buildAssetSpec,
|
|
12
14
|
buildManifest,
|
|
13
15
|
estimateTokens,
|
|
14
16
|
formatResidentCostBlock,
|
|
15
17
|
init_esm_shims,
|
|
16
18
|
residentCost,
|
|
17
19
|
resolveBundleRoot
|
|
18
|
-
} from "./chunk-
|
|
20
|
+
} from "./chunk-3QBHZUVB.js";
|
|
19
21
|
|
|
20
22
|
// src/trust-tier-drift.ts
|
|
21
23
|
init_esm_shims();
|
|
@@ -66,6 +68,7 @@ function classifyDrift(tier, stars) {
|
|
|
66
68
|
export {
|
|
67
69
|
CATEGORIES,
|
|
68
70
|
CLI_BASES,
|
|
71
|
+
DEFAULT_OPTIONS,
|
|
69
72
|
DEV_METHOD_SKILL_IDS,
|
|
70
73
|
EXTERNAL_ASSETS,
|
|
71
74
|
INTERNAL_BUNDLED_SKILL_IDS,
|
|
@@ -74,6 +77,7 @@ export {
|
|
|
74
77
|
TRUST_TIER,
|
|
75
78
|
assetCliSupport,
|
|
76
79
|
assetCostRows,
|
|
80
|
+
buildAssetSpec,
|
|
77
81
|
buildManifest,
|
|
78
82
|
classifyDrift,
|
|
79
83
|
driftTargets,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/trust-tier-drift.ts"],"sourcesContent":["/**\n * A1 — Trust Tier star-drift 검출 데이터 + 순수 로직.\n *\n * TRUST_TIER 의 star 기반 라벨(vetted ≥ 1000★ / experimental < 1000★)이 실제 GitHub\n * star 와 어긋났는지(drift) 판정한다. `official` 은 star 무관(Anthropic 공식·하네스 자체)\n * 이라 검사 제외.\n *\n * repo 출처 = 각 자산 method (in-code authoritative — 주석이 아니라 실제 설치 source):\n * skill → method.source (\"owner/repo\" 또는 github URL)\n * plugin → method.marketplace (\"owner/repo\")\n * npm → NPM_REPO_OVERRIDE[id] (pkg 는 npm 명이므로 GitHub repo 를 별도 명시)\n *\n * fetch/네트워크는 본 모듈에 없음 — 순수 로직만(테스트 가능). 실 fetch 는\n * `scripts/trust-tier-drift.mjs` 가 담당.\n */\nimport { EXTERNAL_ASSETS, type ExternalAsset, TRUST_TIER } from \"./external-assets.js\";\n\n// v26.79.0 — gen-compatibility 의 카테고리 exhaustiveness 가드용 SSOT (하드코딩 drift 차단).\nexport { CATEGORIES } from \"./categories.js\";\n// v26.116.0 (ADR-043) — context-cost-report.mjs 가 dist 에서 비용 계측기를 읽도록 re-export.\n// v26.140.0 — formatResidentCostBlock: 리포트가 표를 직접 조립하지 않도록 표시 계약도 함께 노출.\nexport {\n assetCostRows,\n estimateTokens,\n formatResidentCostBlock,\n residentCost,\n resolveBundleRoot,\n} from \"./context-cost.js\";\n// v26.76.0 — gen-compatibility.mjs 가 dist 에서 자산 카탈로그+tier 를 읽도록 re-export.\n// v26.93.0 — DEV_METHOD_SKILL_IDS 추가: gen-compatibility 의 CLI scope override 를\n// 하드코딩 id 목록 대신 SSOT 에서 derive (no-false-ship drift 구조 차단).\n// v26.95.0 — INTERNAL_BUNDLED_SKILL_IDS (dev-method + opt-in gemini-consult) 로 CLI scope derive.\nexport {\n // v26.102.0 (ADR-031) — gen-compatibility 의 CLI 열이 도달 범위를 derive 하도록 re-export.\n assetCliSupport,\n DEV_METHOD_SKILL_IDS,\n EXTERNAL_ASSETS,\n INTERNAL_BUNDLED_SKILL_IDS,\n TRUST_TIER,\n} from \"./external-assets.js\";\nexport { buildManifest } from \"./manifest.js\";\n// v26.102.0 (ADR-031) — 도달 라벨(\"N-CLI\")의 N 을 derive 하기 위한 re-export (매직 넘버 금지).\nexport { CLI_BASES, TRACKS } from \"./types.js\";\n\n/** vetted 경계 (NORTH_STAR / PRD v26-71 D2). */\nexport const STAR_THRESHOLD = 1000;\n\nexport type StarTier = \"vetted\" | \"experimental\";\nexport type DriftVerdict = \"ok\" | \"promote\" | \"demote\";\n\n/**\n * method 가 GitHub repo 를 안 담는 자산(npm.pkg / npx-run.cmd 는 npm 명) → 트러스트 근거가\n * 된 GitHub repo 를 명시 매핑. override 가 method 도출보다 우선.\n */\nconst REPO_OVERRIDE: Record<string, string> = {\n \"vercel-cli\": \"vercel/vercel\", // npm\n \"netlify-cli\": \"netlify/cli\", // npm\n \"supabase-cli\": \"supabase/cli\", // npm\n \"agent-browser\": \"vercel-labs/agent-browser\", // npm\n openspec: \"Fission-AI/OpenSpec\", // npm (v26.75.0)\n \"bmad-method\": \"bmad-code-org/BMAD-METHOD\", // npx-run (v26.75.0)\n};\n\n/** \"https://github.com/owner/repo\" 또는 \"owner/repo[/...]\" → \"owner/repo\". 실패 시 null. */\nexport function normalizeRepo(source: string): string | null {\n const stripped = source.replace(/^https?:\\/\\/github\\.com\\//i, \"\");\n const m = stripped.match(/^([^/\\s]+\\/[^/\\s]+)/);\n return m?.[1] ?? null;\n}\n\n/** 자산의 GitHub owner/repo 도출. override 우선 → skill/plugin method. 도출 불가 시 null. */\nexport function repoForAsset(asset: ExternalAsset): string | null {\n const override = REPO_OVERRIDE[asset.id];\n if (override) return override;\n const m = asset.method;\n if (m.kind === \"skill\") return normalizeRepo(m.source);\n if (m.kind === \"plugin\") return normalizeRepo(m.marketplace);\n return null;\n}\n\nexport interface DriftTarget {\n id: string;\n tier: StarTier;\n repo: string;\n}\n\n/** star 기반(vetted/experimental) 자산만 + repo 도출 가능한 것만 검사 대상. */\nexport function driftTargets(\n assets: ReadonlyArray<ExternalAsset> = EXTERNAL_ASSETS,\n): DriftTarget[] {\n const out: DriftTarget[] = [];\n for (const a of assets) {\n const tier = TRUST_TIER[a.id];\n if (tier !== \"vetted\" && tier !== \"experimental\") continue;\n const repo = repoForAsset(a);\n if (!repo) continue; // 도출 불가 — 테스트가 0건을 강제하므로 정상 경로에선 발생 안 함\n out.push({ id: a.id, tier, repo });\n }\n return out;\n}\n\n/** 정적 tier 가 실제 star 와 어긋났는지 판정. */\nexport function classifyDrift(tier: StarTier, stars: number): DriftVerdict {\n if (tier === \"vetted\" && stars < STAR_THRESHOLD) return \"demote\";\n if (tier === \"experimental\" && stars >= STAR_THRESHOLD) return \"promote\";\n return \"ok\";\n}\n"],"mappings":"
|
|
1
|
+
{"version":3,"sources":["../src/trust-tier-drift.ts"],"sourcesContent":["/**\n * A1 — Trust Tier star-drift 검출 데이터 + 순수 로직.\n *\n * TRUST_TIER 의 star 기반 라벨(vetted ≥ 1000★ / experimental < 1000★)이 실제 GitHub\n * star 와 어긋났는지(drift) 판정한다. `official` 은 star 무관(Anthropic 공식·하네스 자체)\n * 이라 검사 제외.\n *\n * repo 출처 = 각 자산 method (in-code authoritative — 주석이 아니라 실제 설치 source):\n * skill → method.source (\"owner/repo\" 또는 github URL)\n * plugin → method.marketplace (\"owner/repo\")\n * npm → NPM_REPO_OVERRIDE[id] (pkg 는 npm 명이므로 GitHub repo 를 별도 명시)\n *\n * fetch/네트워크는 본 모듈에 없음 — 순수 로직만(테스트 가능). 실 fetch 는\n * `scripts/trust-tier-drift.mjs` 가 담당.\n */\nimport { EXTERNAL_ASSETS, type ExternalAsset, TRUST_TIER } from \"./external-assets.js\";\n\n// v26.79.0 — gen-compatibility 의 카테고리 exhaustiveness 가드용 SSOT (하드코딩 drift 차단).\nexport { CATEGORIES } from \"./categories.js\";\n// v26.116.0 (ADR-043) — context-cost-report.mjs 가 dist 에서 비용 계측기를 읽도록 re-export.\n// v26.140.0 — formatResidentCostBlock: 리포트가 표를 직접 조립하지 않도록 표시 계약도 함께 노출.\nexport {\n assetCostRows,\n estimateTokens,\n formatResidentCostBlock,\n residentCost,\n resolveBundleRoot,\n} from \"./context-cost.js\";\n// v26.76.0 — gen-compatibility.mjs 가 dist 에서 자산 카탈로그+tier 를 읽도록 re-export.\n// v26.93.0 — DEV_METHOD_SKILL_IDS 추가: gen-compatibility 의 CLI scope override 를\n// 하드코딩 id 목록 대신 SSOT 에서 derive (no-false-ship drift 구조 차단).\n// v26.95.0 — INTERNAL_BUNDLED_SKILL_IDS (dev-method + opt-in gemini-consult) 로 CLI scope derive.\nexport {\n // v26.102.0 (ADR-031) — gen-compatibility 의 CLI 열이 도달 범위를 derive 하도록 re-export.\n assetCliSupport,\n DEV_METHOD_SKILL_IDS,\n EXTERNAL_ASSETS,\n INTERNAL_BUNDLED_SKILL_IDS,\n TRUST_TIER,\n} from \"./external-assets.js\";\nexport { buildAssetSpec, buildManifest } from \"./manifest.js\";\n// v26.102.0 (ADR-031) — 도달 라벨(\"N-CLI\")의 N 을 derive 하기 위한 re-export (매직 넘버 금지).\nexport { CLI_BASES, DEFAULT_OPTIONS, TRACKS } from \"./types.js\";\n\n/** vetted 경계 (NORTH_STAR / PRD v26-71 D2). */\nexport const STAR_THRESHOLD = 1000;\n\nexport type StarTier = \"vetted\" | \"experimental\";\nexport type DriftVerdict = \"ok\" | \"promote\" | \"demote\";\n\n/**\n * method 가 GitHub repo 를 안 담는 자산(npm.pkg / npx-run.cmd 는 npm 명) → 트러스트 근거가\n * 된 GitHub repo 를 명시 매핑. override 가 method 도출보다 우선.\n */\nconst REPO_OVERRIDE: Record<string, string> = {\n \"vercel-cli\": \"vercel/vercel\", // npm\n \"netlify-cli\": \"netlify/cli\", // npm\n \"supabase-cli\": \"supabase/cli\", // npm\n \"agent-browser\": \"vercel-labs/agent-browser\", // npm\n openspec: \"Fission-AI/OpenSpec\", // npm (v26.75.0)\n \"bmad-method\": \"bmad-code-org/BMAD-METHOD\", // npx-run (v26.75.0)\n};\n\n/** \"https://github.com/owner/repo\" 또는 \"owner/repo[/...]\" → \"owner/repo\". 실패 시 null. */\nexport function normalizeRepo(source: string): string | null {\n const stripped = source.replace(/^https?:\\/\\/github\\.com\\//i, \"\");\n const m = stripped.match(/^([^/\\s]+\\/[^/\\s]+)/);\n return m?.[1] ?? null;\n}\n\n/** 자산의 GitHub owner/repo 도출. override 우선 → skill/plugin method. 도출 불가 시 null. */\nexport function repoForAsset(asset: ExternalAsset): string | null {\n const override = REPO_OVERRIDE[asset.id];\n if (override) return override;\n const m = asset.method;\n if (m.kind === \"skill\") return normalizeRepo(m.source);\n if (m.kind === \"plugin\") return normalizeRepo(m.marketplace);\n return null;\n}\n\nexport interface DriftTarget {\n id: string;\n tier: StarTier;\n repo: string;\n}\n\n/** star 기반(vetted/experimental) 자산만 + repo 도출 가능한 것만 검사 대상. */\nexport function driftTargets(\n assets: ReadonlyArray<ExternalAsset> = EXTERNAL_ASSETS,\n): DriftTarget[] {\n const out: DriftTarget[] = [];\n for (const a of assets) {\n const tier = TRUST_TIER[a.id];\n if (tier !== \"vetted\" && tier !== \"experimental\") continue;\n const repo = repoForAsset(a);\n if (!repo) continue; // 도출 불가 — 테스트가 0건을 강제하므로 정상 경로에선 발생 안 함\n out.push({ id: a.id, tier, repo });\n }\n return out;\n}\n\n/** 정적 tier 가 실제 star 와 어긋났는지 판정. */\nexport function classifyDrift(tier: StarTier, stars: number): DriftVerdict {\n if (tier === \"vetted\" && stars < STAR_THRESHOLD) return \"demote\";\n if (tier === \"experimental\" && stars >= STAR_THRESHOLD) return \"promote\";\n return \"ok\";\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAAA;AA6CO,IAAM,iBAAiB;AAS9B,IAAM,gBAAwC;AAAA,EAC5C,cAAc;AAAA;AAAA,EACd,eAAe;AAAA;AAAA,EACf,gBAAgB;AAAA;AAAA,EAChB,iBAAiB;AAAA;AAAA,EACjB,UAAU;AAAA;AAAA,EACV,eAAe;AAAA;AACjB;AAGO,SAAS,cAAc,QAA+B;AAC3D,QAAM,WAAW,OAAO,QAAQ,8BAA8B,EAAE;AAChE,QAAM,IAAI,SAAS,MAAM,qBAAqB;AAC9C,SAAO,IAAI,CAAC,KAAK;AACnB;AAGO,SAAS,aAAa,OAAqC;AAChE,QAAM,WAAW,cAAc,MAAM,EAAE;AACvC,MAAI,SAAU,QAAO;AACrB,QAAM,IAAI,MAAM;AAChB,MAAI,EAAE,SAAS,QAAS,QAAO,cAAc,EAAE,MAAM;AACrD,MAAI,EAAE,SAAS,SAAU,QAAO,cAAc,EAAE,WAAW;AAC3D,SAAO;AACT;AASO,SAAS,aACd,SAAuC,iBACxB;AACf,QAAM,MAAqB,CAAC;AAC5B,aAAW,KAAK,QAAQ;AACtB,UAAM,OAAO,WAAW,EAAE,EAAE;AAC5B,QAAI,SAAS,YAAY,SAAS,eAAgB;AAClD,UAAM,OAAO,aAAa,CAAC;AAC3B,QAAI,CAAC,KAAM;AACX,QAAI,KAAK,EAAE,IAAI,EAAE,IAAI,MAAM,KAAK,CAAC;AAAA,EACnC;AACA,SAAO;AACT;AAGO,SAAS,cAAc,MAAgB,OAA6B;AACzE,MAAI,SAAS,YAAY,QAAQ,eAAgB,QAAO;AACxD,MAAI,SAAS,kBAAkB,SAAS,eAAgB,QAAO;AAC/D,SAAO;AACT;","names":[]}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uzysjung/agent-harness",
|
|
3
|
-
"version": "26.
|
|
3
|
+
"version": "26.151.0",
|
|
4
4
|
"description": "Curate vetted AI-coding skills & plugins by your tech stack — install only what you need, across Claude Code, Codex, OpenCode & Antigravity",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"publishConfig": {
|
package/templates/CLAUDE.md
CHANGED
|
@@ -1,166 +1,147 @@
|
|
|
1
|
-
#
|
|
1
|
+
# CLAUDE.md
|
|
2
2
|
|
|
3
|
-
These are default decision principles
|
|
4
|
-
them.
|
|
3
|
+
These are default decision principles, not a fixed workflow.
|
|
4
|
+
Project-specific policy may refine them. Approval and independent-review
|
|
5
|
+
gates below are mandatory.
|
|
6
|
+
|
|
7
|
+
## 1. Resolve what matters, then act
|
|
8
|
+
|
|
9
|
+
Inspect relevant code, contracts, tests, and worktree changes before editing.
|
|
10
|
+
Expand investigation as needed to understand the change and its risks.
|
|
11
|
+
|
|
12
|
+
Resolve questions from project evidence first. Verify exact external API, CLI,
|
|
13
|
+
authentication, and policy details against the actual environment or applicable
|
|
14
|
+
authoritative sources before relying on them. Reuse current, relevant evidence.
|
|
15
|
+
|
|
16
|
+
For product and planning work, identify the target user's problem, current
|
|
17
|
+
alternatives, and the outcome the core journey should deliver. Assess whether
|
|
18
|
+
the proposed approach is worth choosing over those alternatives and what
|
|
19
|
+
observable evidence would support that judgment. Reuse established context
|
|
20
|
+
and distinguish observed evidence from assumptions or simulated feedback.
|
|
21
|
+
Test unresolved assumptions that could change direction through the smallest
|
|
22
|
+
useful research or prototype before costly commitments.
|
|
23
|
+
|
|
24
|
+
Ask before committing to an unresolved choice with material consequences that
|
|
25
|
+
would be costly to reverse; explain the meaningful options and trade-offs.
|
|
26
|
+
Otherwise, choose a reasonable interpretation and continue; state assumptions
|
|
27
|
+
that affect the result.
|
|
28
|
+
|
|
29
|
+
Investigate unexpected results before proposing another fix. Do not stack
|
|
30
|
+
speculative fixes without updating the diagnosis.
|
|
31
|
+
|
|
32
|
+
## 2. Choose the simplest sufficient solution
|
|
33
|
+
|
|
34
|
+
Choose the least complex solution that fully satisfies the requested outcome.
|
|
35
|
+
For consequential choices, compare existing solutions, proven patterns, and
|
|
36
|
+
credible alternatives within scope; skip formal comparisons when the choice
|
|
37
|
+
is clear. Use abstractions and local refactoring when they simplify the
|
|
38
|
+
solution; do not optimize merely for fewer lines or a smaller diff.
|
|
39
|
+
|
|
40
|
+
For service and substantial feature work, prefer small, end-to-end increments
|
|
41
|
+
that exercise the core user journey and expose risky assumptions or integrations
|
|
42
|
+
early. Optimize for time to a verified, usable outcome, including likely rework,
|
|
43
|
+
not just time to the first implementation. Continue until the agreed scope is
|
|
44
|
+
complete.
|
|
45
|
+
|
|
46
|
+
Include the behavior necessary to make the requested capability usable and
|
|
47
|
+
correct. Do not add unrequested features or speculative extension points.
|
|
48
|
+
Add defensive logic for concrete requirements, credible failure modes, and
|
|
49
|
+
trust boundaries.
|
|
50
|
+
|
|
51
|
+
If the requested approach conflicts with its goal or constraints, explain the
|
|
52
|
+
trade-off and recommend a better option without silently changing scope.
|
|
53
|
+
|
|
54
|
+
## 3. Keep changes focused and preserve existing work
|
|
55
|
+
|
|
56
|
+
Change what the task and its verification require. Leave unrelated cleanup
|
|
57
|
+
alone, match local style, and remove only artifacts made obsolete by your change.
|
|
58
|
+
|
|
59
|
+
Preserve existing contracts and intentional behavior unless changing them is
|
|
60
|
+
part of the request. Security requirements take precedence over local convention.
|
|
61
|
+
|
|
62
|
+
Do not overwrite, revert, stage, or reformat pre-existing user changes without
|
|
63
|
+
explicit authorization. If overlapping changes prevent safe editing, report
|
|
64
|
+
the conflict and stop only the affected work.
|
|
65
|
+
|
|
66
|
+
## 4. Define success and verify proportionally
|
|
67
|
+
|
|
68
|
+
Define observable completion criteria and suitable verification before editing.
|
|
69
|
+
Base them on the requested outcome, intended use, relevant user journey and
|
|
70
|
+
core behavior, constraints, and material risks. Distinguish required readiness
|
|
71
|
+
from optional polish; do not silently lower the former or expand the latter.
|
|
72
|
+
For complex or risky work, share a short plan. Routine changes do not require
|
|
73
|
+
a formal planning document.
|
|
74
|
+
|
|
75
|
+
Use checks that demonstrate the required behavior and cover material risks.
|
|
76
|
+
Prefer regression tests for reproducible bug fixes and behavior changes.
|
|
77
|
+
When automation is impractical, use the strongest feasible alternative and
|
|
78
|
+
report its limits.
|
|
79
|
+
|
|
80
|
+
For runnable changes, execute the relevant behavior through focused tests,
|
|
81
|
+
direct execution, or both, as needed to demonstrate the completion criteria,
|
|
82
|
+
in an authorized target or representative environment. Inspect the result
|
|
83
|
+
and fix failures; report required execution checks that cannot be performed
|
|
84
|
+
within scope.
|
|
85
|
+
|
|
86
|
+
For UI changes, inspect the rendered result and test affected interactions and
|
|
87
|
+
states. Assess usability in the relevant supported layouts against the
|
|
88
|
+
completion criteria and the existing or agreed design.
|
|
89
|
+
|
|
90
|
+
Run the applicable required checks. Once the completion criteria and required
|
|
91
|
+
checks are satisfied, repeat or expand verification only when changes, failures,
|
|
92
|
+
or unresolved risks warrant it. Do not weaken criteria or bypass required checks
|
|
93
|
+
to claim success.
|
|
94
|
+
|
|
95
|
+
Independent review by an agent that did not author the work is required before
|
|
96
|
+
adopting a spec, plan, or design artifact as a basis for downstream work, before
|
|
97
|
+
declaring an implementation complete, and before deployment.
|
|
98
|
+
|
|
99
|
+
Routine execution notes do not need separate review unless they introduce
|
|
100
|
+
material decisions not already reviewed. Scale review depth to the change's
|
|
101
|
+
impact and risk; small, low-risk changes need only a focused review.
|
|
102
|
+
|
|
103
|
+
Give the reviewer the original request, constraints, completion criteria, actual
|
|
104
|
+
artifacts, and verification evidence. The reviewer must assess both the criteria
|
|
105
|
+
and the work, not merely the author's summary. Blocking findings are unmet
|
|
106
|
+
required criteria or substantiated, material risks to correctness, security,
|
|
107
|
+
data integrity, or usability. Resolve them with fixes or evidence before
|
|
108
|
+
proceeding. Separate optional improvements and preferences from blockers.
|
|
109
|
+
|
|
110
|
+
Review applies to the reviewed artifact version and context. Reuse it while
|
|
111
|
+
both remain applicable; re-review affected areas when changes or new evidence
|
|
112
|
+
invalidate it. Review does not replace execution checks. If independent review
|
|
113
|
+
is unavailable, stop at the affected gate and report it; self-review does not
|
|
114
|
+
satisfy the gate.
|
|
115
|
+
|
|
116
|
+
## 5. Report evidence and stop unproductive loops
|
|
117
|
+
|
|
118
|
+
Report what changed, the evidence for completed criteria, and relevant remaining
|
|
119
|
+
gaps. Do not present unverified work as complete. Distinguish required checks
|
|
120
|
+
from optional broader checks; not running an optional check is not itself a
|
|
121
|
+
blocker.
|
|
122
|
+
|
|
123
|
+
When retries stop producing new evidence, stop the failing approach and provide
|
|
124
|
+
a precise blocker and handoff rather than continuing blindly.
|
|
125
|
+
|
|
126
|
+
## 6. Keep authority explicit
|
|
127
|
+
|
|
128
|
+
Work autonomously within the authorized scope. Within existing approvals, carry
|
|
129
|
+
the task through implementation, applicable execution checks, and fixes without
|
|
130
|
+
pausing for routine confirmation. At a gate, stop only dependent actions and
|
|
131
|
+
continue authorized work that does not require crossing it.
|
|
132
|
+
|
|
133
|
+
Beyond required reviews, delegate independent tasks when the expected time or
|
|
134
|
+
quality benefit outweighs coordination cost. Parallelize implementation only
|
|
135
|
+
with non-overlapping ownership and clear interfaces. Keep delegated work within the
|
|
136
|
+
same scope and authority; own the integrated result.
|
|
137
|
+
|
|
138
|
+
Before destructive or privileged actions, deployment, or shared-state writes,
|
|
139
|
+
require explicit approval covering the action and target unless that approval
|
|
140
|
+
already exists. A general objective is not approval.
|
|
141
|
+
|
|
142
|
+
Ordinary local edits and cleanup of your own disposable artifacts within scope
|
|
143
|
+
do not need separate approval. This does not authorize discarding pre-existing
|
|
144
|
+
user work or data.
|
|
5
145
|
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
Before editing, inspect the affected code, tests, callers, interfaces,
|
|
9
|
-
dependencies, documentation, and worktree changes. Resolve questions from the
|
|
10
|
-
repository before asking the user.
|
|
11
|
-
|
|
12
|
-
Before designing, examine how established products solve the same problem.
|
|
13
|
-
Prefer proven patterns. Verify external behavior, specifications, failure
|
|
14
|
-
modes, and library capabilities from current authoritative sources; do not
|
|
15
|
-
guess. When only an outside source can answer and you cannot reach one, say
|
|
16
|
-
which question is unanswered rather than filling it in.
|
|
17
|
-
|
|
18
|
-
State uncertainty plainly and distinguish facts, assumptions, and judgments.
|
|
19
|
-
If an unresolved choice could materially affect behavior, data, security,
|
|
20
|
-
cost, architecture, or scope and would be expensive to reverse, present the
|
|
21
|
-
options and trade-offs and ask before proceeding. When independent lanes
|
|
22
|
-
disagree or the call is genuinely uncertain, settle it with an adversarial
|
|
23
|
-
panel of independent reviewers rather than the loudest lane; a panel costs
|
|
24
|
-
more than a decision that is cheap to undo is worth. Otherwise, state a
|
|
25
|
-
reasonable assumption and continue.
|
|
26
|
-
|
|
27
|
-
Mention a simpler sufficient approach when one exists. Push back when a request
|
|
28
|
-
conflicts with the goal, contract, or security boundary.
|
|
29
|
-
|
|
30
|
-
## 2. Define Success and Keep It Simple
|
|
31
|
-
|
|
32
|
-
Before editing, define observable completion criteria and how each will be
|
|
33
|
-
verified. For multi-step work, use a short plan with verification points.
|
|
34
|
-
|
|
35
|
-
Prefer regression tests at stable contract boundaries. If automated testing is
|
|
36
|
-
impractical, state why and define the strongest reproducible alternative.
|
|
37
|
-
|
|
38
|
-
Implement the minimum change that completely satisfies the request. Do not add
|
|
39
|
-
unrequested features, speculative configuration, one-use abstractions,
|
|
40
|
-
unnecessary indirection, unused extension points, or defensive code without a
|
|
41
|
-
credible failure mode, contract, trust boundary, or security requirement.
|
|
42
|
-
|
|
43
|
-
Prefer direct, explicit, reproducible, and testable behavior. If equally
|
|
44
|
-
sufficient approaches exist, choose the simplest one that reaches a verified
|
|
45
|
-
result soonest. Brevity is not simplicity when it obscures behavior or
|
|
46
|
-
verification.
|
|
47
|
-
|
|
48
|
-
When building something that does not exist yet, start with the smallest
|
|
49
|
-
working end-to-end path and add one verified capability at a time. Do not trade
|
|
50
|
-
working code for unfinished complexity.
|
|
51
|
-
|
|
52
|
-
## 3. Preserve Sound Boundaries
|
|
53
|
-
|
|
54
|
-
Separate modules only where responsibilities, trust boundaries, lifecycle, or
|
|
55
|
-
reasons to change differ. Keep interfaces narrow; do not abstract hypothetical
|
|
56
|
-
reuse.
|
|
57
|
-
|
|
58
|
-
Before implementing or adding a package, inspect installed dependencies and
|
|
59
|
-
verify their versions, documentation, types, and capabilities. Prefer
|
|
60
|
-
maintained libraries when they reduce total complexity or improve reliability.
|
|
61
|
-
Do not reimplement common functionality without a concrete reason.
|
|
62
|
-
|
|
63
|
-
Make architectural decisions for the system's expected lifetime. Avoid both
|
|
64
|
-
speculative generality and temporary designs known to require replacement.
|
|
65
|
-
|
|
66
|
-
Do not preserve backward compatibility unless an active contract or persisted
|
|
67
|
-
data requires it. Delete verified-unused paths instead of adding compatibility
|
|
68
|
-
layers, fallbacks, dual paths, or migrations. A path counts as verified-unused
|
|
69
|
-
only when every caller you found is inside this repository; when a consumer can
|
|
70
|
-
be outside it, you cannot establish that from here. Breaking active
|
|
71
|
-
dependencies requires explicit authorization.
|
|
72
|
-
|
|
73
|
-
## 4. Make Surgical Changes
|
|
74
|
-
|
|
75
|
-
Change only what the request and its verification require. Do not refactor,
|
|
76
|
-
reformat, rename, rewrite, or delete unrelated code. Remove only artifacts made
|
|
77
|
-
obsolete by the change or paths verified as unused and safe to remove.
|
|
78
|
-
|
|
79
|
-
Leave unrelated dead code untouched. Report it only if it materially affects
|
|
80
|
-
the task or verification.
|
|
81
|
-
|
|
82
|
-
Follow local style unless it conflicts with a contract, security boundary,
|
|
83
|
-
data integrity, or intentionally tested behavior.
|
|
84
|
-
|
|
85
|
-
Pre-existing changes belong to the user. Do not overwrite, revert, stage, or
|
|
86
|
-
reformat them. Stop if they overlap the target and safe editing is unclear.
|
|
87
|
-
|
|
88
|
-
## 5. Verify and Review
|
|
89
|
-
|
|
90
|
-
Run targeted checks first, then broaden according to risk. Iterate until the
|
|
91
|
-
completion criteria pass. Do not weaken or silently omit criteria. If blocked,
|
|
92
|
-
report exactly what remains unmet and why.
|
|
93
|
-
|
|
94
|
-
Independent review by an agent or person other than the one that produced the
|
|
95
|
-
work is required at two points: for a completed specification, plan, or design
|
|
96
|
-
before it is built on, and for any completed change before it is merged into
|
|
97
|
-
shared work.
|
|
98
|
-
|
|
99
|
-
Give the reviewer the completion criteria and relevant constraints. A reviewer
|
|
100
|
-
verifies the work itself rather than trusting the author's report, so
|
|
101
|
-
independent review supplements direct verification; it does not replace it. At
|
|
102
|
-
these boundaries, an unreviewed artifact is not verified. Starting a review is
|
|
103
|
-
always available, so "no reviewer" is a decision rather than a condition: if
|
|
104
|
-
you proceed without one, the artifact stays unverified — say so, and never
|
|
105
|
-
present self-review as independent review.
|
|
106
|
-
|
|
107
|
-
## 6. Protect High-Impact Boundaries
|
|
108
|
-
|
|
109
|
-
Before any destructive, privileged, costly, or shared-state operation, state
|
|
110
|
-
the exact action and target and obtain explicit approval. Do not infer approval
|
|
111
|
-
from a broad objective.
|
|
112
|
-
|
|
113
|
-
Preparing a migration, deployment, release, command, or other reviewable
|
|
114
|
-
artifact does not authorize applying it to shared or persistent state.
|
|
115
|
-
|
|
116
|
-
These principles shape decisions; they do not block actions. Anything that must
|
|
117
|
-
hold every time regardless of judgment belongs in the enforcement layer, not in
|
|
118
|
-
a sentence here.
|
|
119
|
-
|
|
120
|
-
## 7. Report Evidence
|
|
121
|
-
|
|
122
|
-
Report what changed, what was verified and how, what independent review found,
|
|
123
|
-
what was not verified, what remains, and the risk that remains.
|
|
124
|
-
|
|
125
|
-
Do not claim `Pass`, `Works`, or `Completed` without evidence. An unverified
|
|
126
|
-
criterion is incomplete. Disclose relevant broader checks not run; their
|
|
127
|
-
absence does not invalidate separately verified results.
|
|
128
|
-
|
|
129
|
-
If repeated attempts produce no new evidence, stop and provide a concise
|
|
130
|
-
handoff.
|
|
131
|
-
|
|
132
|
-
## Presenting a decision
|
|
133
|
-
|
|
134
|
-
Present a decision or approval request as AS-IS → TO-BE with a recommendation
|
|
135
|
-
and the trade-off, not as prose.
|
|
136
|
-
|
|
137
|
-
**Write it from the position of whoever lives with the result** — the person who
|
|
138
|
-
uses what you are building, or the operator who runs it. Name that role, and say
|
|
139
|
-
what they can do now that they could not before, or what stops happening to them;
|
|
140
|
-
a field added to a module is not something anyone outside the code can feel. When
|
|
141
|
-
a change has no user-visible effect, say who does benefit rather than inventing a
|
|
142
|
-
user.
|
|
143
|
-
|
|
144
|
-
Give the surrounding before/after context in enough detail that the reader does
|
|
145
|
-
not have to ask, and show the choice the way they will meet it — a comparison
|
|
146
|
-
table, a sketch, a rendered example — rather than describing it. When the reader
|
|
147
|
-
says they don't follow, fix what the words point at before rewording; the usual
|
|
148
|
-
cause is one name meaning two things.
|
|
149
|
-
|
|
150
|
-
## Skills that apply continuously
|
|
151
|
-
|
|
152
|
-
A skill's body loads when the prompt looks like the skill's job. That is enough
|
|
153
|
-
for task-shaped skills and not enough for these, which apply to every response
|
|
154
|
-
or every delegation — nothing in a prompt ever looks like those, so without a
|
|
155
|
-
line here they never open. Each is selected individually at install time, hence
|
|
156
|
-
the condition on every line.
|
|
157
|
-
|
|
158
|
-
- `clear-korean-communication`, where installed — applies to every answer,
|
|
159
|
-
report, and approval request, including the AS-IS → TO-BE form above; not
|
|
160
|
-
only at the moment approval is asked for.
|
|
161
|
-
- `task-brief`, where installed — normalize an incoming work request into the
|
|
162
|
-
brief shape before starting, fill the fields it left open from context, and
|
|
163
|
-
show the filled-in brief so the user can carry it straight into a prompt,
|
|
164
|
-
marking which values were assumed.
|
|
165
|
-
- `model-orchestration`, where installed — when work is delegated, it decides
|
|
166
|
-
which lane takes the work and how that lane is run.
|
|
146
|
+
Preparing a migration, deployment change, or other reviewable artifact does not
|
|
147
|
+
authorize applying it to shared systems or persistent application data.
|
|
@@ -104,7 +104,7 @@ npx eslint . --fix
|
|
|
104
104
|
## When NOT to Use
|
|
105
105
|
|
|
106
106
|
- Code needs refactoring or new features → use the `implementer` agent
|
|
107
|
-
- Security issues →
|
|
107
|
+
- Security issues → run Claude Code's `/security-review` on the diff
|
|
108
108
|
|
|
109
109
|
---
|
|
110
110
|
|
|
@@ -44,7 +44,7 @@ origin: self-authored (GSD gsd-plan-checker 사상 흡수, 100% 자체 작성)
|
|
|
44
44
|
- "Phase 2는 Phase 1 완료 후" 같은 명시적 순서가 있는지 확인.
|
|
45
45
|
|
|
46
46
|
### D5. Context Budget
|
|
47
|
-
- SPEC.md
|
|
47
|
+
- SPEC.md 가 길어져 한 화면에 안 들어오면 기능별 or 영역별 분리를 제안(WARNING).
|
|
48
48
|
- plan.md에 30개 이상 task가 한 Phase에 몰려 있으면 WARNING (분해 필요).
|
|
49
49
|
- 각 task의 예상 파일 수 × 평균 크기가 context window의 50% 초과 시 WARNING.
|
|
50
50
|
|
|
@@ -12,11 +12,12 @@ context: fork
|
|
|
12
12
|
|
|
13
13
|
당신은 **검증자**다. 구현자가 아니다. 생성자 관점을 완전히 배제하고, 까다로운 리뷰어 관점에서만 평가하라.
|
|
14
14
|
|
|
15
|
-
Anthropic Harness Design 연구의 핵심 발견: "생성(generator)과 평가(evaluator)를 분리하면 품질이 비약적으로 향상된다."
|
|
16
|
-
|
|
17
15
|
## Review Process
|
|
18
16
|
|
|
19
17
|
### Step 1: Context Gathering
|
|
18
|
+
|
|
19
|
+
**입력 = 사용자 씬과 그 완료 기준.** diff 는 그 씬의 범위에서 읽는다 — 씬을 이루는 변경분을 모아 한 번에 본다. 씬과 완료 기준을 받지 못했으면 판정 전에 요청자에게 먼저 묻는다.
|
|
20
|
+
|
|
20
21
|
```bash
|
|
21
22
|
git diff --staged
|
|
22
23
|
git diff
|
|
@@ -36,9 +37,7 @@ git log --oneline -10
|
|
|
36
37
|
|
|
37
38
|
#### Readability (가독성)
|
|
38
39
|
- 함수/변수 이름이 의도를 드러내는가?
|
|
39
|
-
-
|
|
40
|
-
- 파일 길이 ≤ 800줄인가?
|
|
41
|
-
- 중첩 깊이 ≤ 4레벨인가?
|
|
40
|
+
- 함수·파일 길이, 중첩 깊이가 **이 저장소의 관례에 비해** 눈에 띄게 큰가? (절대 숫자가 아니라 주변 코드와 비교한다 — 관례를 따르는 코드에 리팩터를 요구하지 않는다)
|
|
42
41
|
- 불필요한 주석 없이 코드 자체가 설명적인가?
|
|
43
42
|
|
|
44
43
|
#### Architecture (아키텍처)
|
|
@@ -1,9 +1,5 @@
|
|
|
1
1
|
# {PROJECT_NAME} — Antigravity Agent Guide
|
|
2
2
|
|
|
3
|
-
> **Generated from**: `templates/CLAUDE.md` (4-CLI 단일 원본) via TS CLI `src/antigravity/transform.ts`
|
|
4
|
-
> **Antigravity**: 2.0+ (`agy` CLI + desktop IDE)
|
|
5
|
-
> **Location**: workspace rule — `.agents/rules/uzys-harness.md`
|
|
6
|
-
|
|
7
3
|
## Project Context
|
|
8
4
|
|
|
9
5
|
{PROJECT_CONTEXT}
|
|
@@ -12,23 +8,7 @@
|
|
|
12
8
|
|
|
13
9
|
{PROJECT_RULES}
|
|
14
10
|
|
|
15
|
-
## Protected Files
|
|
16
|
-
|
|
17
|
-
- `.env*`, `**/credentials.json`
|
|
18
|
-
- `*.lock`, `package-lock.json`, `pnpm-lock.yaml`, `poetry.lock`, `Cargo.lock`, `uv.lock`
|
|
19
|
-
- `.git/` 내부 파일 (커밋 메시지/hook 제외)
|
|
20
|
-
- `~/.gemini/`, `~/.claude/`, `~/.codex/` 글로벌 (D16 보호)
|
|
21
|
-
|
|
22
|
-
보호 영역 이슈 발견 시 **보고만**. 직접 수정 금지.
|
|
23
|
-
|
|
24
|
-
## Scopes
|
|
25
|
-
|
|
26
|
-
| Scope | 위치 | 비고 |
|
|
27
|
-
|-------|------|------|
|
|
28
|
-
| Workspace skills | `.agents/skills/` | 본 프로젝트 한정 (Codex 공유) |
|
|
29
|
-
| Workspace rules | `.agents/rules/` | 본 문서 |
|
|
30
|
-
| Global rules | `~/.gemini/GEMINI.md` | 사용자 직접 관리 (harness 미터치) |
|
|
31
|
-
|
|
32
|
-
---
|
|
11
|
+
## Protected Files
|
|
33
12
|
|
|
34
|
-
|
|
13
|
+
- lock 파일(`package-lock.json` · `pnpm-lock.yaml` · `poetry.lock` · `Cargo.lock` · `uv.lock` 등)은 **손으로 고치지 않는다** — 패키지 매니저로 재생성한다.
|
|
14
|
+
- `.env*` · `**/credentials.json` · `.git/` 내부(커밋 메시지·hook 제외) · `~/.gemini/` `~/.claude/` `~/.codex/` 전역 설정은 **보고만** 한다. 직접 수정하지 않는다.
|
|
@@ -1,9 +1,5 @@
|
|
|
1
1
|
# {PROJECT_NAME} — Codex Agent Guide
|
|
2
2
|
|
|
3
|
-
> **Generated from**: `templates/CLAUDE.md` (4-CLI 단일 원본) via `scripts/claude-to-codex.sh` (Phase C)
|
|
4
|
-
> **Codex Version**: 0.124.0+
|
|
5
|
-
> **Linked SPEC**: `docs/specs/codex-compat.md`
|
|
6
|
-
|
|
7
3
|
## Project Context
|
|
8
4
|
|
|
9
5
|
{PROJECT_CONTEXT}
|
|
@@ -25,58 +21,11 @@
|
|
|
25
21
|
|
|
26
22
|
## Session Start
|
|
27
23
|
|
|
28
|
-
|
|
29
|
-
1. `docs/SPEC.md` 및 `docs/specs/*.md` 재참조 (Persistent Anchor)
|
|
30
|
-
2. `docs/todo.md` 현재 Phase 확인
|
|
31
|
-
|
|
32
|
-
`session_start` hook이 자동 수행. Hook 실패 시 수동 수행.
|
|
33
|
-
|
|
34
|
-
## Protected Files (DO NOT EDIT)
|
|
35
|
-
|
|
36
|
-
Codex `sandbox_mode = "workspace-write"` + `approval_policy = "on-request"` 가 1차 방어. LLM 추가 준수:
|
|
37
|
-
|
|
38
|
-
- `.env*`
|
|
39
|
-
- `**/credentials.json`
|
|
40
|
-
- `*.lock`, `package-lock.json`, `pnpm-lock.yaml`, `poetry.lock`, `Cargo.lock`, `uv.lock`
|
|
41
|
-
- `.git/` 내부 파일 (커밋 메시지/hook 제외)
|
|
42
|
-
- `~/.codex/`, `~/.claude/` 글로벌 (D16 보호)
|
|
43
|
-
|
|
44
|
-
보호 영역 이슈 발견 시 **보고만**. 직접 수정 금지.
|
|
45
|
-
|
|
46
|
-
## Git Policy
|
|
47
|
-
|
|
48
|
-
- 코드/문서 변경 시 **즉시 commit**. "나중에 한꺼번에" 금지.
|
|
49
|
-
- `main` 직접 커밋 금지. feature branch 사용.
|
|
50
|
-
- Conventional Commits — `<type>: <description>` (feat, fix, refactor, docs, test, chore, perf, ci)
|
|
51
|
-
|
|
52
|
-
## Agents (subagent, multi_agent stable)
|
|
53
|
-
|
|
54
|
-
| Agent | Model | 역할 |
|
|
55
|
-
|-------|-------|------|
|
|
56
|
-
| reviewer | opus | 검증 전용 (SOD). 5축 리뷰 |
|
|
57
|
-
| data-analyst | opus | Python / DuckDB / Trino / ML / PySide6 |
|
|
58
|
-
| strategist | opus | 제안서 / DD / PPT / 경쟁분석 / 재무모델 |
|
|
59
|
-
| code-reviewer | sonnet | 일상적 코드 리뷰 |
|
|
60
|
-
| security-reviewer | sonnet | OWASP Top 10, 보안 패턴 |
|
|
61
|
-
|
|
62
|
-
subagent 호출은 `spawn_agent / wait_agent / close_agent` 툴. 부모 컨텍스트 격리.
|
|
63
|
-
|
|
64
|
-
## Hooks 현황 (Codex 0.124.0 실측 제약)
|
|
65
|
-
|
|
66
|
-
- `pre_tool_use` / `post_tool_use` — **Bash 툴 한정 발화** (Issue #16732). ApplyPatch(파일 쓰기) 가로채기는 불가. `sandbox_mode` + `approval_policy`로 대체 보호.
|
|
67
|
-
- 프로젝트 `.codex/config.toml` 훅은 사용자 `~/.codex/config.toml`에 trust entry 등록된 경우에만 로드.
|
|
68
|
-
- 인터랙티브 세션 hook 로딩 bug (Issue #17532) 존재 가능 — `codex exec` 비대화형은 정상.
|
|
69
|
-
|
|
70
|
-
## Experience Accumulation
|
|
71
|
-
|
|
72
|
-
- Codex `memories` feature (experimental) — Claude auto memory 유사
|
|
73
|
-
- 검증된 learning만 Rules 승격
|
|
74
|
-
|
|
75
|
-
## Context Management
|
|
24
|
+
`docs/SPEC.md` 가 있으면 세션 시작 때 먼저 읽는다 — 현재 범위와 완료 기준의 앵커다. 없으면 이 줄은 해당 없다.
|
|
76
25
|
|
|
77
|
-
|
|
78
|
-
- `child_agents_md` feature flag는 **under development, disabled** — AGENTS.md 디렉토리 계층 merge 사용 불가. 글로벌 `~/.codex/AGENTS.md` + 프로젝트 `AGENTS.md` 2단만 사용.
|
|
26
|
+
## Protected Files
|
|
79
27
|
|
|
80
|
-
|
|
28
|
+
Codex `sandbox_mode = "workspace-write"` + `approval_policy = "on-request"` 가 1차 방어다. 그 위에:
|
|
81
29
|
|
|
82
|
-
|
|
30
|
+
- lock 파일(`package-lock.json` · `pnpm-lock.yaml` · `poetry.lock` · `Cargo.lock` · `uv.lock` 등)은 **손으로 고치지 않는다** — 패키지 매니저로 재생성한다.
|
|
31
|
+
- `.env*` · `**/credentials.json` · `.git/` 내부(커밋 메시지·hook 제외) · `~/.codex/` `~/.claude/` 전역 설정은 **보고만** 한다. 직접 수정하지 않는다.
|
|
@@ -44,6 +44,10 @@ BASENAME=$(basename "$FILE_PATH")
|
|
|
44
44
|
|
|
45
45
|
# 보호 패턴 확인
|
|
46
46
|
case "$BASENAME" in
|
|
47
|
+
# 예시·템플릿 파일에는 시크릿이 없다 — 에이전트가 만들고 고치는 것이 정상이다 (2차 감사 G-01).
|
|
48
|
+
.env.example|.env.sample|.env.template)
|
|
49
|
+
exit 0
|
|
50
|
+
;;
|
|
47
51
|
.env|.env.*)
|
|
48
52
|
log_block "$FILE_PATH"
|
|
49
53
|
echo "BLOCKED: Protected file: $BASENAME. Environment files must be edited manually." >&2
|
|
@@ -27,9 +27,63 @@ fi
|
|
|
27
27
|
# 실패 모드이기 때문이다(cli-development.md §Cross-Platform).
|
|
28
28
|
ORPHAN_NOTE=""
|
|
29
29
|
PROJ_DIR=$(pwd)
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
30
|
+
# **물리 경로도 함께 본다.** macOS 의 `/var` 는 `/private/var` 로의 심링크라 `pwd` 는
|
|
31
|
+
# `/var/...`, 커널·`lsof` 는 `/private/var/...` 를 낸다. 논리 경로만 대면 접두사가 어긋나
|
|
32
|
+
# 실재하는 고아를 **0건으로** 보고한다 — 실제로 그렇게 짰다가 진짜 고아를 만든 시험에서 잡혔다.
|
|
33
|
+
PROJ_DIR_P=$(pwd -P 2>/dev/null || printf '%s' "$PROJ_DIR")
|
|
34
|
+
|
|
35
|
+
# **한 번만 찍고 그 스냅샷을 읽는다.** `ps | grep <이름>` 은 grep 자신의 커맨드라인을 매치해
|
|
36
|
+
# 항상 양성을 낸다 — 이 리포가 실제로 오판한 형태다.
|
|
37
|
+
PS_SNAP=$(ps -eo pid,ppid,command 2>/dev/null || true)
|
|
38
|
+
|
|
39
|
+
# ⓐ 커맨드라인에 프로젝트 경로가 **든** 고아 (`npm run dev /path/...` 처럼 경로를 인자로 받은 것).
|
|
40
|
+
ORPHANS=$(printf '%s\n' "$PS_SNAP" | awk -v d="$PROJ_DIR" '$2==1 && index($0,d)>0 {n++} END{print n+0}')
|
|
41
|
+
|
|
42
|
+
# ⓑ 커맨드라인에 경로가 **없는** 고아. 서브에이전트가 이 모양이다 — 경로는 cwd 에만 있어서
|
|
43
|
+
# ⓐ 는 이 부류를 **구조적으로 0건**으로 보고했다(#326 실측: 살아 있는 서브에이전트 2건 → 0건).
|
|
44
|
+
# 0 을 내는 탐지기는 없는 것보다 나쁘다. 거짓 안심을 준다.
|
|
45
|
+
#
|
|
46
|
+
# **비용 때문에 후보를 먼저 좁힌다.** cwd 를 묻는 것은 macOS 에서 pid 당 약 4 ms 다(실측).
|
|
47
|
+
# ppid=1 전체(이 머신 433개)에 물으면 세션 시작이 배로 느려지고, 그 비용은 설치받은
|
|
48
|
+
# 사람이 **매 세션** 낸다. 그래서 ⓐ 가 이미 세지 않은 것 중 **에이전트 부류만** 보고,
|
|
49
|
+
# 후보가 0이면 cwd 를 아예 안 묻는다. 상한 40개.
|
|
50
|
+
#
|
|
51
|
+
# **비용의 원인을 한 번 잘못 짚었다.** 처음엔 후보 수 탓인 줄 알고 범위를 좁혔는데, 재보니
|
|
52
|
+
# **후보 1개짜리 `lsof` 가 1,444 ms** 였다 — 비싼 것은 후보 수가 아니라 `lsof` 자신이었다.
|
|
53
|
+
# `-b`(블록 가능 호출 회피)를 붙이자 같은 답에 **74 ms** 가 됐고, 그래서 범위를 다시 넓힐 수
|
|
54
|
+
# 있었다. 좁힌 채로 뒀으면 경로가 argv 에 없는 비-에이전트 고아(디렉터리 안에서 띄운
|
|
55
|
+
# `node server.js` 등)를 영영 못 봤을 것이다.
|
|
56
|
+
CAND=$(printf '%s\n' "$PS_SNAP" | awk -v d="$PROJ_DIR" \
|
|
57
|
+
'$2==1 && index($0,d)==0 && /claude|node|python|bun|deno|--agent-name/ {print $1}' | head -40)
|
|
58
|
+
CWD_ORPHANS=0
|
|
59
|
+
if [ -n "$CAND" ]; then
|
|
60
|
+
if [ -d /proc ]; then
|
|
61
|
+
# Linux: /proc 는 사실상 공짜다.
|
|
62
|
+
for pid in $CAND; do
|
|
63
|
+
cwd=$(readlink "/proc/$pid/cwd" 2>/dev/null || true)
|
|
64
|
+
case "$cwd" in
|
|
65
|
+
"$PROJ_DIR" | "$PROJ_DIR"/* | "$PROJ_DIR_P" | "$PROJ_DIR_P"/*) CWD_ORPHANS=$((CWD_ORPHANS + 1)) ;;
|
|
66
|
+
esac
|
|
67
|
+
done
|
|
68
|
+
elif command -v lsof > /dev/null 2>&1; then
|
|
69
|
+
# macOS: /proc 이 없다. 한 번에 묻고, 접두사 일치로 **이 프로젝트 밑**만 센다 —
|
|
70
|
+
# 다른 프로젝트의 고아를 남의 세션이 판단해선 안 된다.
|
|
71
|
+
# `-b` 가 이 검사를 쓸 수 있게 만든다 — 블록 가능한 커널 호출을 피한다. 실측: 후보 1개에
|
|
72
|
+
# `-b` 없이 **1,444 ms**, 있으면 **74 ms**(19배)이고 답은 같다(cwd 28건 동일). `-w` 는 그때
|
|
73
|
+
# 나는 경고를 죽이고, `-n`·`-P` 는 우리가 안 쓰는 이름 해석을 건너뛴다.
|
|
74
|
+
CWD_ORPHANS=$(lsof -b -w -n -P -p "$(printf '%s' "$CAND" | tr '\n' ',' | sed 's/,$//')" -a -d cwd -Fn 2>/dev/null \
|
|
75
|
+
| awk -v d="$PROJ_DIR" -v dp="$PROJ_DIR_P" '
|
|
76
|
+
substr($0,1,1)=="n" {
|
|
77
|
+
p = substr($0,2)
|
|
78
|
+
if (p==d || index(p, d "/")==1 || p==dp || index(p, dp "/")==1) n++
|
|
79
|
+
} END {print n+0}')
|
|
80
|
+
fi
|
|
81
|
+
# lsof 도 /proc 도 없으면 CWD_ORPHANS 는 0 이다. 이 경우는 **못 본 것**이지 없는 것이 아니다.
|
|
82
|
+
fi
|
|
83
|
+
|
|
84
|
+
ORPHAN_TOTAL=$((${ORPHANS:-0} + ${CWD_ORPHANS:-0}))
|
|
85
|
+
if [ "$ORPHAN_TOTAL" -gt 0 ]; then
|
|
86
|
+
ORPHAN_NOTE=" WARNING: ${ORPHAN_TOTAL} orphaned process(es) from a previous session still belong to this project (parent died, reparented to init); ${CWD_ORPHANS} of them are only visible by working directory, which is how subagents look. Inspect with: ps -eo pid,ppid,etime,command | grep \$(pwd) | grep -v grep — and for the rest, resolve each candidate's cwd. Then stop what you recognise. Leaving them costs memory and can hold ports or file locks."
|
|
33
87
|
fi
|
|
34
88
|
|
|
35
89
|
# 3. 세션 컨텍스트 출력
|