agent-bios 0.19.1 → 0.19.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/DEPENDENCIES.md +58 -30
  2. package/INSTALL.md +4 -4
  3. package/README.md +105 -28
  4. package/claude/CLAUDE.md +1 -1
  5. package/claude/guides/claude-prompting.md +1 -1
  6. package/claude/guides/cli-multi-model-workflow.md +4 -4
  7. package/claude/guides/coding-staged-workflow.md +17 -0
  8. package/claude/guides/documentation-hygiene.md +3 -0
  9. package/claude/guides/gpt-prompting.md +1 -1
  10. package/claude/guides/korean-writing.md +153 -0
  11. package/claude/guides/learning-flow.md +4 -4
  12. package/claude/guides/llm-capability-boundary.md +7 -1
  13. package/claude/guides/session-distill-workflow.md +8 -8
  14. package/claude/guides/slide-writing/RUNBOOK.md +5 -5
  15. package/claude/guides/tooling-gotchas.md +20 -1
  16. package/claude/guides/ui-design/visual-direction.md +88 -0
  17. package/claude/guides/ui-design.md +90 -0
  18. package/claude/guides/verification-discipline.md +10 -1
  19. package/claude/hooks/tooling-gotchas-hook.py +41 -0
  20. package/claude/skills/repo-charter/SKILL.md +3 -3
  21. package/claude/skills/understand/SKILL.md +5 -5
  22. package/codex/AGENTS.md +1 -1
  23. package/codex/guides/claude-prompting.md +1 -1
  24. package/codex/guides/cli-multi-model-workflow.md +4 -4
  25. package/codex/guides/coding-staged-workflow.md +17 -0
  26. package/codex/guides/documentation-hygiene.md +3 -0
  27. package/codex/guides/gpt-prompting.md +1 -1
  28. package/codex/guides/korean-writing.md +153 -0
  29. package/codex/guides/learning-flow.md +4 -4
  30. package/codex/guides/llm-capability-boundary.md +7 -1
  31. package/codex/guides/session-distill-workflow.md +8 -8
  32. package/codex/guides/slide-writing/RUNBOOK.md +5 -5
  33. package/codex/guides/tooling-gotchas.md +20 -1
  34. package/codex/guides/ui-design/visual-direction.md +88 -0
  35. package/codex/guides/ui-design.md +90 -0
  36. package/codex/guides/verification-discipline.md +10 -1
  37. package/compose/app_bridge/SKILL.md +12 -12
  38. package/compose/app_bridge/scripts/bridge.py +35 -10
  39. package/compose/app_desktop/server.py +250 -0
  40. package/compose/assemble.py +5 -5
  41. package/compose/bootstrap/SKILL.md +18 -18
  42. package/compose/canary.sh +4 -4
  43. package/compose/check-domains.py +6 -6
  44. package/compose/corpus-state.py +16 -1168
  45. package/compose/corpus.py +13 -402
  46. package/compose/corpus_app.py +14 -450
  47. package/compose/corpus_catalog.py +15 -926
  48. package/compose/corpus_import.py +14 -523
  49. package/compose/corpus_install.py +14 -1847
  50. package/compose/corpus_session.py +16 -848
  51. package/compose/corpus_setup.py +16 -672
  52. package/compose/corpus_setup_cli.py +15 -580
  53. package/compose/corpus_setup_i18n.py +20 -324
  54. package/compose/corpus_setup_ui.py +18 -645
  55. package/compose/corpus_store.py +16 -1664
  56. package/compose/corpus_transaction.py +15 -284
  57. package/compose/corpus_ui.py +17 -972
  58. package/compose/corpus_ui_runtime.py +16 -274
  59. package/compose/corpus_understand.py +13 -671
  60. package/compose/domains.json +3 -1
  61. package/compose/host_platform.py +121 -0
  62. package/compose/instructions-state.py +1178 -0
  63. package/compose/instructions.py +409 -0
  64. package/compose/instructions_app.py +697 -0
  65. package/compose/instructions_catalog.py +931 -0
  66. package/compose/instructions_import.py +537 -0
  67. package/compose/instructions_install.py +1932 -0
  68. package/compose/instructions_session.py +852 -0
  69. package/compose/instructions_setup.py +713 -0
  70. package/compose/instructions_setup_cli.py +607 -0
  71. package/compose/instructions_setup_i18n.py +327 -0
  72. package/compose/instructions_setup_ui.py +647 -0
  73. package/compose/instructions_store.py +1668 -0
  74. package/compose/instructions_transaction.py +308 -0
  75. package/compose/instructions_ui.py +975 -0
  76. package/compose/instructions_ui_runtime.py +279 -0
  77. package/compose/instructions_understand.py +678 -0
  78. package/compose/native_cli.py +52 -0
  79. package/compose/register-hooks.py +1 -1
  80. package/compose/runtime_entry.py +58 -0
  81. package/compose/setup/START.md +11 -11
  82. package/compose/windows_deploy.py +719 -0
  83. package/docs/advanced-launch.md +11 -11
  84. package/docs/instructions-compatibility.md +86 -0
  85. package/docs/{corpus.md → instructions.md} +36 -8
  86. package/docs/recovery.md +10 -10
  87. package/docs/releases/0.19.2.md +38 -0
  88. package/docs/releases/0.19.3.md +107 -0
  89. package/docs/session-model.md +31 -20
  90. package/docs/setup.md +63 -26
  91. package/docs/understand.md +6 -6
  92. package/docs/windows.md +99 -0
  93. package/install.sh +71 -69
  94. package/launch/agent-launch.py +309 -293
  95. package/launch/agent-launch.toml +2 -2
  96. package/launch/agent-launch.zsh +11 -1
  97. package/launch/i18n/en.toml +55 -55
  98. package/launch/i18n/ja.toml +56 -56
  99. package/launch/i18n/ko.toml +56 -56
  100. package/launch/shell_integration.py +4 -4
  101. package/learn/collect-learning.py +10 -10
  102. package/learn/learning.schema.json +1 -1
  103. package/learn/migrate-learnings.py +51 -51
  104. package/package.json +33 -12
  105. package/provenance.json +1 -1
  106. /package/docs/assets/{corpus-studio.svg → instructions-studio.svg} +0 -0
@@ -0,0 +1,153 @@
1
+ ---
2
+ guide_id: korean-writing
3
+ language: en
4
+ status: active
5
+ description: Before writing, revising, or translating Korean prose, you must read and apply this entire guide, including for responses, documents, slide wording, and UI copy.
6
+ use_when:
7
+ - writing, revising, or translating Korean responses or documents
8
+ - composing Korean slide text, headlines, buttons, or status labels
9
+ core_rules:
10
+ - Consult the full guide before Korean writing and compare the result with the source and actual state.
11
+ - Connect the central judgment to evidence while preserving distinct concepts, conditions, and uncertainty.
12
+ - Distinguish proposals, available actions, and completed states; match action copy to actual behavior.
13
+ ---
14
+
15
+ # Writing in Korean
16
+
17
+ Before writing, revising, or translating Korean prose, read and apply this entire guide.
18
+ If the same complete text is already in the current task context, no duplicate file read is needed.
19
+ Apply it to responses, reports, notices, slide wording, and UI copy while respecting the requested
20
+ length, format, and register. Do not force a headline or lengthy evidence onto a short response.
21
+ Preserve quotations and identifiers that must remain exact.
22
+
23
+ First decide what the reader needs to understand or judge. Connect the supporting evidence and
24
+ conditions, and preserve conceptual distinctions and factual scope when shortening the text.
25
+ The Korean golden sentences below are hypothetical teaching examples, not real facts or fixed
26
+ templates. Apply their preservation of meaning, rather than copying sentence counts, headings,
27
+ or endings.
28
+
29
+ ## 1. Define the question and central judgment
30
+
31
+ Make each fact, comparison, and explanation's role in the central judgment clear. Split independent
32
+ questions, or explain why they belong together.
33
+
34
+ - Avoid: “문의가 늘었다. 담당자는 세 명이다. 안내 문서를 개편했다.”
35
+ - Golden: “반복 문의를 줄이기 위해 안내 문서를 개편했다. 담당자 세 명이 자주 받는 질문을 모아 답변을 보강했다.”
36
+
37
+ Use this connection only when the purpose and work described in the golden are supported.
38
+ Do not invent purpose or activities from a list of facts alone.
39
+
40
+ ## 2. Make the subject and central judgment visible in the headline
41
+
42
+ Avoid references that require earlier text or headlines that only preview a count. The argument
43
+ should connect when headlines are read alone. Do not force a conclusion onto an overview,
44
+ definition, or transition.
45
+
46
+ - Avoid: “세 가지 개선 사항”
47
+ - Golden: “신청 절차를 줄여 사용자의 입력 부담을 낮춘다”
48
+
49
+ For a definition, a role-revealing title such as “신청 자격의 정의” is also appropriate.
50
+
51
+ ## 3. Keep claims within the evidence's scope and certainty
52
+
53
+ Distinguish hypotheses from confirmed facts, examples from actual selections, and temporal order
54
+ from causality. Retain uncertainty where the source has not established a relationship.
55
+
56
+ - Avoid: “안내 문서 개편으로 문의가 감소했다.”
57
+ - Golden: “안내 문서 개편 후 문의가 감소했다. 다만 같은 기간 이용자 수도 줄어, 개편의 효과인지는 확인되지 않았다.”
58
+
59
+ Mention the decline in users only when supported too. Do not invent another fact to explain an
60
+ unconfirmed cause.
61
+
62
+ ## 4. Use the same name for the same concept and distinguish different concepts
63
+
64
+ Do not alternate synonyms merely for stylistic variety. Even when shortening an explanation,
65
+ retain each independent concept's name, definition, and difference from others.
66
+
67
+ - Avoid: “활성 사용자는 주간 이용자를 뜻한다. 참여 고객은 이번 주 120명이다.”
68
+ - Golden: “활성 사용자는 일주일 동안 한 번 이상 서비스를 이용한 사용자다. 이번 주 활성 사용자는 120명이다.”
69
+
70
+ Use the source or agreed definition. Ambiguous terminology does not authorize a new threshold.
71
+
72
+ ## 5. Preserve the subject and action even in short wording
73
+
74
+ Qualify ambiguous words such as scope, criteria, or completion with the object the reader needs.
75
+ Do not fill gaps in compressed wording with meaning absent from the source.
76
+
77
+ - Avoid: “통과 범위와 보완 항목을 함께 남깁니다.”
78
+ - Golden: “검토를 통과한 범위와 보완할 항목을 함께 기록합니다.”
79
+
80
+ Name each state the reader must distinguish directly.
81
+
82
+ - Avoid: “0처럼 보이는 누락을 구분합니다.”
83
+ - Golden: “값이 누락된 경우와 실제 금액이 0인 경우를 구분합니다.”
84
+
85
+ ## 6. State relationships between concepts explicitly
86
+
87
+ Make clear what causes, conditions, or forms part of what, and what is compared with what.
88
+ Do not add unsupported causality, sequence, or superiority to create a connection.
89
+
90
+ - Avoid: “교육 참여와 배포 권한은 연결됩니다.”
91
+ - Golden: “교육 이수는 배포 권한을 신청하기 위한 조건입니다. 교육을 이수해도 권한이 자동으로 부여되지는 않습니다.”
92
+
93
+ ## 7. Move from judgment to evidence
94
+
95
+ Present the central judgment and necessary premises, then the supporting explanation. Distinguish
96
+ new implications derived from the explanation and avoid repeating the same content in several places.
97
+
98
+ - Avoid: “연동 시험 두 건이 남았다. 금요일에 시험 환경을 사용할 수 있다. 출시 일정 조정이 필요하다.”
99
+ - Golden: “출시를 다음 주로 미뤄야 한다. 필수 연동 시험 두 건이 남아 있으며, 시험 환경은 이번 주 금요일부터 사용할 수 있다.”
100
+
101
+ This example assumes the release timing and mandatory tests are established. Do not settle a
102
+ schedule or condition absent from the source.
103
+
104
+ ## 8. Keep conditions, exceptions, and scope close to the claim
105
+
106
+ Do not relegate interpretation-changing conditions to incidental information. Align units, periods,
107
+ subjects, and denominators when comparing numbers; disclose differences in the comparison bases.
108
+
109
+ - Avoid: “모든 사용자는 신청을 취소할 수 있다.”
110
+ - Golden: “사용자는 승인 전까지 신청을 취소할 수 있다. 승인 후에는 담당자에게 취소를 요청해야 한다.”
111
+
112
+ ## 9. Distinguish proposals, available actions, and completed states
113
+
114
+ Do not describe a proposed procedure as an implemented feature. Name what has completed and
115
+ keep review, approval, finalization, and transmission as distinct states.
116
+
117
+ - Avoid, on a proposal screen before implementation: “원천 자료부터 회계 시스템 입력용 집계까지 검토합니다.”
118
+ - Golden: “원천 자료부터 회계 시스템 입력용 집계까지, 검토 절차를 제안합니다.”
119
+
120
+ - Avoid: “검토가 완료되어 회계 처리가 끝났습니다.”
121
+ - Golden: “검토를 완료했습니다. 회계 승인과 결산 확정 여부는 별도로 확인해야 합니다.”
122
+
123
+ ## 10. Match action copy to actual behavior
124
+
125
+ Buttons and links should say what the user will do or see. Distinguish viewing, selecting, saving,
126
+ and submitting. Do not promise a result that the click alone does not achieve.
127
+
128
+ - Avoid, on a button opening a scope explanation: “검토 시작”
129
+ - Golden: “검토 범위 안내 보기”
130
+
131
+ If the destination actually allows selection, use “검토 기간·상품 선택”.
132
+
133
+ ## 11. Remove repetition while retaining necessary explanation
134
+
135
+ Do not delete essential evidence, definitions, or conditions for brevity. Do not add claims or
136
+ repeat statements to fill space. Separate sentences with different roles, such as definition
137
+ and interpretation.
138
+
139
+ - Avoid: “처리 시간을 단축하고 더 빠르게 처리하기 위해 중복 확인 절차를 없애 처리 속도를 개선한다.”
140
+ - Golden: “처리 시간을 줄이기 위해 같은 정보를 두 번 확인하는 절차를 한 번으로 합친다.”
141
+
142
+ ## 12. Compare the finished text with the source and actual state
143
+
144
+ Check that key concepts, figures, conditions, subjects, and relationships survive. For features
145
+ and procedures, also verify available behavior and current state. The headlines and body should
146
+ communicate the argument without the author's additional explanation.
147
+
148
+ - Source: “시범 운영에 참여한 20개 팀 중 12개 팀이 다음 분기에도 사용할 의향이 있다고 답했다.”
149
+ - Avoid: “고객의 60%가 재계약을 확정했다.”
150
+ - Golden: “시범 운영에 참여한 20개 팀 중 12개 팀(60%)이 다음 분기에도 사용할 의향을 밝혔다.”
151
+
152
+ Preserve the subject, denominator, and response meaning when summarizing. Do not turn intent
153
+ to use into a confirmed renewal.
@@ -76,7 +76,7 @@ Surface each surviving candidate compactly — lesson, type, intended layer,
76
76
  admission-bar verdict, domain (+ proposed_domain) — and record ONLY what the
77
77
  user explicitly approves.
78
78
 
79
- In a Codex app task using the explicit corpus bridge, use the `learn` command
79
+ In a Codex app task using the explicit instructions bridge, use the `learn` command
80
80
  and environment returned in its `runtime` metadata, or its registered helper.
81
81
  Shell exports from an earlier app tool call do not persist into later calls.
82
82
 
@@ -96,11 +96,11 @@ match your host):
96
96
 
97
97
  The script (capability boundary) owns `learning_id` / `created` / `schema_version`
98
98
  and validates against `learn/learning.schema.json`. It appends the record to the
99
- private corpus's `learnings/<host>/events.jsonl`, keeping Claude and Codex captures
99
+ private instruction store's `learnings/<host>/events.jsonl`, keeping Claude and Codex captures
100
100
  separate. Selected learnings enter future activated-session snapshots through the
101
- private corpus. Capture preserves the user's global instruction files and the
101
+ private instructions. Capture preserves the user's global instruction files and the
102
102
  running session's snapshot. Upload runs after local storage when transport is configured.
103
- The private root is `$AGENT_BIOS_CORPUS_DIR`, defaulting to
103
+ The private root is `$AGENT_BIOS_INSTRUCTIONS_DIR`, defaulting to
104
104
  `~/.config/agent-bios/corpus`; native host-home settings do not relocate it.
105
105
  `--config-dir` is restricted to explicit legacy mode. Use `--no-upload` for
106
106
  local-only capture or `--dry-run` to validate without writes or uploads.
@@ -110,7 +110,13 @@ Important levers:
110
110
  grounding checks, provenance checks, citation checks, static checks, E2E
111
111
  checks, and semantic quality gates.
112
112
  - Retry/fail policy: retry transient generation failures; fail clearly when the
113
- available route cannot enforce the required contract.
113
+ available route cannot enforce the required contract. A contract-failing item
114
+ is rejected loudly and never recorded as a valid result; record its failed or
115
+ not-done status in artifact state. On a production path the whole run halts
116
+ only when continuing would contaminate what the next step reads — a stale or
117
+ not-reusable input, a contract-failing value about to be recorded as valid,
118
+ or an external write with an unknown outcome; otherwise it warns loudly and
119
+ continues, and the next run picks up the work not done.
114
120
  - Observability: prompt packet snapshot, model/provider version, schema hash,
115
121
  source snapshot, validator decision, retry reason, and artifact lineage.
116
122
 
@@ -6,7 +6,7 @@ audience: author
6
6
  use_when:
7
7
  - a session was launched with the Session distill preset (mission-injected)
8
8
  - the launcher nudge says enough sessions accumulated for a mining window
9
- - learning from LLM work sessions to improve the corpus and its application
9
+ - learning from LLM work sessions to improve the instructions and its application
10
10
  - promoting, incubating, or retiring items in the session-distill ledger
11
11
  core_rules:
12
12
  - read Goal and desired outcomes before state files or pipeline work; use it to judge the run and its delegated work
@@ -69,7 +69,7 @@ keep existing content, or a clearly bounded unresolved finding, is also useful.
69
69
  Carry this goal and the relevant outcome criteria into delegated work, then
70
70
  assess its results against them before presenting the run as complete.
71
71
 
72
- **Requires an agent-bios checkout.** This runbook edits the corpus itself, so it
72
+ **Requires an agent-bios checkout.** This runbook edits the instructions themselves, so it
73
73
  names repo paths and runs repo scripts. On a packaged install those do not exist:
74
74
  say so and stop rather than following steps you cannot execute.
75
75
 
@@ -87,7 +87,7 @@ Everything durable lives in the agent-bios repo.
87
87
  correct on the day it is written and silently wrong afterwards.
88
88
  2. `design/session-distill/versions.json` — authoring provenance mapping each
89
89
  closed mining window to its commit. Private rollback selects an installed
90
- `baseline_ref` through the corpus plan; this registry is not that authority.
90
+ `baseline_ref` through the instructions plan; this registry is not that authority.
91
91
  3. `design/session-distill/PLACEMENT-FRAMEWORK.md` — the placement framework
92
92
  (typology A–G, layers, admission bars, lifecycle). Apply the current
93
93
  `AGENTS.md` reductions-only rule and `SURFACES.md` delivery contract when
@@ -105,7 +105,7 @@ Run in order; each stage reads the previous stage's `out/`:
105
105
  2. `digest.py` — one secret-redacted digest per session with deterministic
106
106
  6-criteria signals. Screen ALL digests; triage orders, never drops.
107
107
  3. `batch.py` — the baseline blob (`claude/CLAUDE.md` + every guide, the
108
- repo's canonical corpus) and per-provider batches; writes
108
+ repo's canonical instructions) and per-provider batches; writes
109
109
  `out/batch_index.json`, which the screeners take as their `args`.
110
110
  4. Provider-affine screening against that baseline: `screen-claude.js`
111
111
  (Claude sessions; a Workflow script — pass the index as `args`, one
@@ -173,12 +173,12 @@ Run in order; each stage reads the previous stage's `out/`:
173
173
  dated corrections for anything refuted.
174
174
  2. Write a new timestamped completion record under `design/session-distill/`
175
175
  following `AGENTS.md`; incidental finds become next-window candidates.
176
- 3. Register the corpus version: append {version = window end, commit = the
177
- corpus-close commit} to `design/session-distill/versions.json` for authoring provenance. Private rollback selects an installed `baseline_ref`
178
- through `agent-bios corpus plan`; the window registry does not authorize a
176
+ 3. Register the instructions version: append {version = window end, commit = the
177
+ instructions-close commit} to `design/session-distill/versions.json` for authoring provenance. Private rollback selects an installed `baseline_ref`
178
+ through `agent-bios instructions plan`; the window registry does not authorize a
179
179
  global-file rollback. Then run
180
180
  `python3 session-distill/update-state.py --window-end <date>`
181
- (nudge baseline) and `corpus-state.py project` (launcher status panel).
181
+ (nudge baseline) and `instructions-state.py project` (launcher status panel).
182
182
  4. Merge the branch, push, and confirm the private release from this checkout
183
183
  (`bash install.sh verify`); report stored-state and session-delivery evidence
184
184
  separately.
@@ -13,7 +13,7 @@ Do not substitute an HTML deliverable or claim that unrelated screenshots passed
13
13
  this paired runtime.
14
14
 
15
15
  Use `<runbook-root>` for the directory containing this file and its `scripts/`
16
- directory, and `<job>` for a new directory outside the immutable corpus bundle.
16
+ directory, and `<job>` for a new directory outside the immutable instructions bundle.
17
17
  The runtime refuses an existing job directory; revised inputs or output use a
18
18
  new job revision. In every command, `--base` names the companion directory. The
19
19
  criteria source is its sibling, `<runbook-root>/../slide-writing.md`.
@@ -36,8 +36,8 @@ python3 -B "<runbook-root>/scripts/pair.py" --base "<runbook-root>" prepare \
36
36
  ```
37
37
 
38
38
  Omit `--asset` when no assets are needed; repeat it for additional files. Asset
39
- basenames must be unique. They are copied to `input/assets/<name>`, so URLs from
40
- `output/deck.html` use `../input/assets/<name>`.
39
+ basenames must be unique. They are copied to `<job>/input/assets/<name>`, so URLs from
40
+ `<job>/output/deck.html` use `../input/assets/<name>`.
41
41
 
42
42
  `check` parses and validates the primary criteria source without writing to the
43
43
  guide bundle. `prepare` derives `<job>/input/slide-writing.md` and the job-only
@@ -116,8 +116,8 @@ python3 -B "<runbook-root>/scripts/pair.py" --base "<runbook-root>" verify --job
116
116
  ```
117
117
 
118
118
  Every consuming command also performs its own preflight checks. An edit becomes
119
- available to the next activated corpus snapshot and the next prepared job. An
120
- existing job verifies against its original immutable corpus snapshot, frozen
119
+ available to the next activated instructions snapshot and the next prepared job. An
120
+ existing job verifies against its original immutable instructions snapshot, frozen
121
121
  criterion source, derived oracle, and runtime version. Mutating the guide or code
122
122
  path recorded by that job instead of using its original snapshot invalidates the
123
123
  binding, as do changes to its source document, specification, assets, HTML,
@@ -256,7 +256,26 @@ depends on it, pin it explicitly instead of trusting the environment.
256
256
  and check for a front-side cache/CDN separately. Aim the probe at an
257
257
  in-unit sentinel the app answers without credentials: a denial the app
258
258
  produces anyway passes with the control off, and a redirect into it is a
259
- bypass.
259
+ bypass. The address an IP-based rule compares is the one selected at its
260
+ enforcement point, and configuration alone may not establish which: behind
261
+ a proxy or CDN it can be the client address the configured forwarded-header
262
+ trust chain hands the app, and a managed platform can route particular
263
+ destinations outside an otherwise configured NAT path — on GCP, Google API
264
+ traffic did not use the Cloud NAT address despite
265
+ `privateIpGoogleAccess: false`. Before writing or editing an allowlist or
266
+ perimeter rule, observe the address for the real workload and destination
267
+ at the enforcement point, with a control path that should show a different
268
+ one.
269
+ - **Locating a credential must not print it**: a command meant only to find
270
+ where a secret lives, or to check that it is set, exposes the value if it
271
+ emits it into the transcript — `cat` of an env file, `printenv`,
272
+ `echo $TOKEN`, a keychain read with `-w`, or a grep whose match is the
273
+ secret line. Check existence, length, or shape without emitting any secret
274
+ characters (`[ -n "${X:-}" ]`, `wc -c < file`, a key-name-only listing, a
275
+ nonprinting format check), and let the consumer read the value directly
276
+ from the environment or credential store only when it needs it. Do not
277
+ print even a partial prefix. A value that reached the transcript is
278
+ exposed: rotate it.
260
279
  - **Tightening exposure is a behavior change for external clients**: switching
261
280
  ingress mode, adding an allowlist, or requiring auth is not safe when
262
281
  callers live outside your redeploy. Enumerate which clients reach the
@@ -0,0 +1,88 @@
1
+ ---
2
+ language: en
3
+ status: active
4
+ ---
5
+
6
+ # Choosing a visual direction for an operational interface
7
+
8
+ Use when a new visual direction or a material arrangement decision remains
9
+ unresolved in the requested work. For an established direction or a bounded copy
10
+ correction, use the existing system and verify the affected result proportionately.
11
+
12
+ ## Start from the task and supported system
13
+
14
+ Identify the content people must read, compare, select, edit or act on. Preserve
15
+ the working scope, authority and state distinctions that matter to that task.
16
+ Inspect the project's existing components and tokens before introducing a new
17
+ visual vocabulary. An external reference does not itself justify replacing the
18
+ project's framework or component library.
19
+
20
+ Consult a reference when it helps resolve the remaining visual decision; reuse
21
+ sufficient existing evidence. Research more when the decision needs it or the
22
+ user requests research. Prefer task-relevant product views or official pattern
23
+ documentation. Distinguish live observations, release screenshots, historical
24
+ images and explanatory artwork. Record the source and date, what it informs,
25
+ and its applicability limits in the ordinary design note. Do not infer token
26
+ values or usability from an image alone.
27
+
28
+ ## Choose only the patterns the task needs
29
+
30
+ | Task | Useful arrangement | Meaning to preserve |
31
+ | --- | --- | --- |
32
+ | Reading and review | Give the artifact sufficient space; keep necessary evidence nearby when switching would obstruct comparison | The artifact, source, applicable conditions and review target |
33
+ | Repeated record comparison | Align comparable values; distinguish whole-list, selected-set and individual-record actions | Units, comparison basis, exceptions and each action's scope |
34
+ | Selection-driven editing | Keep relevant properties bound to the selected object | Selection, draft and resulting changes remain attached to the same target |
35
+ | Input and result confirmation | Keep input, consequential effects and results close enough to follow | Entered, saved, approved and actually executed states stay distinct |
36
+
37
+ These are conditional patterns, not a required navigation structure or panel
38
+ count. Choose persistent supporting panels or transient overlays according to
39
+ their role. Give prose a readable measure and comparisons sufficient width.
40
+ Adapt the arrangement to the supported viewport while preserving the current
41
+ selection, return context, reading order and keyboard focus order. Compact
42
+ navigation when needed to keep the main task reachable on a narrow screen.
43
+
44
+ ## Define roles before values
45
+
46
+ Explain the proposed reading order, emphasis, density and grouping briefly.
47
+ Map foregrounds, surfaces, action states, spacing relationships and any required
48
+ depth to the existing semantic tokens. Keep focus, selection, feedback and
49
+ disabled states distinct even when values initially coincide. Add a pattern
50
+ token only when it represents an independent decision. The role of a token
51
+ must survive theme changes; check its actual foreground/background pairing.
52
+
53
+ For information-dense operational work, start with typography, alignment and
54
+ spacing to organize content; add containers, color or elevation when they
55
+ communicate a needed relationship or interaction. Large introductory areas,
56
+ repeated summary cards, badges and icon backgrounds need a task or brand purpose.
57
+ Ordinary content need not look raised. Use depth when it clarifies a floating,
58
+ movable or otherwise distinct object. Preserve useful brand and user preferences.
59
+
60
+ Do not impose a reference palette, fixed pixel scale, permanently dim navigation,
61
+ universal card ban or identical layout across tasks. When no system exists, mark
62
+ initial values as project proposals. Define consistent text roles and spacing
63
+ within and between groups, then check the resulting density. Resolve crowded
64
+ content through grouping, widths and wrapping before shrinking text. Keep words
65
+ and meaningful phrases together where the language permits, with a fallback for
66
+ unbroken strings. Check representative Korean text when the product uses Korean.
67
+
68
+ ## Compare only enough to resolve the decision
69
+
70
+ Use representative content, long labels, important exceptions and relevant
71
+ states at the target sizes. Keep task, content and capability constant while
72
+ comparing visual alternatives; use a credible baseline. Distinguish token
73
+ changes from changes to regions, information order and grouping. Typography
74
+ and spacing can still alter wrapping and geometry. Where narrow layouts
75
+ converge, do not claim a difference the user cannot see.
76
+
77
+ Inspect the actual rendered result when appearance is being judged. Check
78
+ contrast, clipping, wrapping, state distinctions, action targets and focus
79
+ visibility under the applicable accessibility requirements. Verify that
80
+ rearrangement preserves semantic reading and keyboard order. For interactive
81
+ changes, exercise selection, draft retention and the relevant failure/recovery
82
+ paths; styling does not establish permission or execution.
83
+
84
+ Record visual preference separately from observed task performance. Rendered
85
+ samples support visual inspection; interaction and user-performance claims need
86
+ their own evidence. Preserve the result and remaining uncertainty in ordinary
87
+ team artifacts. No prescribed number of variants or new research report is
88
+ needed for every task.
@@ -0,0 +1,90 @@
1
+ ---
2
+ guide_id: ui-design
3
+ language: en
4
+ status: active
5
+ description: For designing, reviewing, or changing task flows, information layout, visual hierarchy, or interaction in operational user interfaces; keep bounded corrections proportionate.
6
+ use_when:
7
+ - designing or reviewing a work application, operations tool, or interactive work surface
8
+ - changing information arrangement, visual hierarchy, design tokens, or interaction states
9
+ - preserving scope, evidence, authority, and continuity through a user interface
10
+ ---
11
+
12
+ # Operational interface design and delivery
13
+
14
+ Use for interfaces where people inspect evidence, compose work, make decisions
15
+ or act on records. Match the deliverable to the user's requested stage and scope.
16
+ A clear text correction needs that correction and a proportionate check.
17
+
18
+ 1. **Establish what the sources mean now.** Separate implemented behavior,
19
+ accepted requirements, proposed design choices and illustrative states.
20
+ Reconcile amendments and decisions before reusing an older screen. A newer
21
+ date alone does not establish authority. Keep material contradictions and
22
+ unknowns visible instead of silently choosing a convenient interpretation.
23
+
24
+ 2. **Preserve requested coverage; bound the work at the right level.** Identify
25
+ the operator, outcome and evidence that would complete this deliverable.
26
+ Distinguish people consuming work context from those authoring or managing
27
+ it; do not invent their visit frequency or force a common starting screen.
28
+ For a complete design request, cover the required operations even when their
29
+ backend is not built; specify their subjects, inputs, effects, dependencies
30
+ and recovery rather than claiming implementation. For a bounded change,
31
+ complete the requested path before expanding neighboring features. Include
32
+ adjacent repairs when that path depends on them or this change causes a
33
+ regression, and state why. An unavailable implementation does not erase a
34
+ requested design contract.
35
+
36
+ 3. **Organize around a meaningful next decision.** Show the working scope,
37
+ evidence to compare, current state, available action and resulting next step
38
+ together. Internal modules or protocol phases are not automatically menus
39
+ or buttons. Combine preparation steps behind one authorized intent when no
40
+ user choice is needed, while keeping distinct outcomes understandable. Use
41
+ the team's visual language to assign emphasis according to the task, and
42
+ make typography, spacing, color and boundaries express meaningful
43
+ relationships. Test whether actual task content is prominent at the target
44
+ size; decorative summaries must earn their space. When layout uncertainty
45
+ matters, compare arrangements at the lowest useful fidelity, explaining the
46
+ task tradeoff and what evidence would change the choice.
47
+
48
+ When a new visual direction or a material arrangement decision remains
49
+ unresolved, use
50
+ `${CLAUDE_CONFIG_DIR:-$HOME/.claude}/guides/ui-design/visual-direction.md`.
51
+ An established direction or a bounded correction does not require new
52
+ reference research or multiple designs.
53
+
54
+ 4. **Keep qualifications with the thing they qualify.** Preserve applicable
55
+ conditions, exceptions, units, source/version, time meaning and affected
56
+ scope through summaries, edits, comparisons and exports. Keep independent
57
+ dimensions separate. Missing evidence is not an empty result or a zero, and
58
+ an index or relationship view is not proof of its underlying body or meaning.
59
+ Expand technical detail when useful without hiding a consequential limit.
60
+
61
+ 5. **Make each action's authority and effect precise.** Name the exact target,
62
+ base and changed content when reviewing a change. Recheck a changed base or
63
+ permission instead of carrying approval onto different input. Distinguish
64
+ selection, saved draft, approval, execution and recipient use as applicable;
65
+ one positive state does not prove the next. Preserve request identity when
66
+ an outcome is uncertain and reconcile its actual result before retrying.
67
+ UI controls reflect the existing authoritative policy; they do not create it.
68
+ If policy is unsettled, label proposed choices and the responsible decision
69
+ owner rather than implying permission or omitting the required design.
70
+
71
+ 6. **Carry the same work across views and people.** Preserve the draft, target,
72
+ review/request identity and return context when moving between surfaces.
73
+ Equivalent outcomes need suitable representations, not identical layouts or
74
+ duplicated business rules. Provide keyboard and structured alternatives for
75
+ necessary spatial interactions. Verify evidence for the actual recipient;
76
+ another worker's receipt does not establish their access or delivery.
77
+
78
+ 7. **Verify this deliverable, then finish it.** Walk a design's concrete scenario
79
+ and relevant exceptions against its sources; inspect its proposed composition.
80
+ For working changes, exercise the changed runtime path, relevant failures,
81
+ keyboard and recovery. When a visual change is material, inspect the affected
82
+ composition with representative content and relevant states at the target
83
+ size, distinguishing token changes from changes to information arrangement.
84
+ Check that visual, reading and focus order remain coherent across layouts.
85
+ Report visual preference separately from observed task performance. Keep
86
+ simulation, observed runtime and user evidence distinct. Once the required
87
+ checks are complete, leave the result, material decisions, source bindings
88
+ and remaining work in ordinary team artifacts. Keep private facts within
89
+ their permitted boundary. Choose dependencies only when a requirement
90
+ demands a decision, using the supported environment first.
@@ -77,6 +77,7 @@ cheapest one to write.
77
77
  - A/B or on/off measurements: before accepting a null result, verify the arms actually received different treatment in the mechanism under test — a shared default or unconditional upstream step can silently apply the treatment to both arms.
78
78
  - Multi-stage pipelines with nondeterministic stages: a final-output diff cannot attribute an effect or a regression to a stage — it conflates the change with run-to-run variance. Persist every stage's output, tabulate what each creates, may edit, and only guards, restrict the suspects to the stages that edit the content in question, and find the first stage where the intended effect disappears or the defect appears. Fix there, preferring a structural recheck over another prompt-level instruction that already failed.
79
79
  - Before/after comparisons: pin the input to an immutable copy — a snapshot or versioned artifact — and run both arms against it, because a live artifact (a growing log, a regenerated upstream stage) drifts between runs and any diff over it, a matching one included, is evidence of nothing; when the arms are metered, restore the baseline's exact upstream inputs and re-run only the changed stage. This is input identity, not the separate unit-and-denominator basis rule.
80
+ - Cost figures from provider usage records: before pricing, map every provider's token fields onto one schema — uncached input, cache read, cache write, output. Some providers report input as a total that already includes cached tokens: treating that total as uncached input and then adding the cache-read field again counts the cached portion twice, while charging the inclusive total once at the full rate misprices that portion instead. Other providers exclude cached tokens from the input field. When a provider reports or prices cache writes separately, keep them in their own field. Confirm each provider's field meaning against its own usage documentation or a known sample, never by analogy with another provider, and derive the cache hit rate from the normalized fields.
80
81
  - Model-behavior guardrails: verify by changed behavior, not recitation — a staged battery from named-trigger cases through disguised, deconfounded, category-wide, and single-variable framings; a clean pass means "no known defect", so re-run the battery when the model changes.
81
82
  - Branch/version test builds against real data: explicitly separate every state sink the app touches (files, DB, OS-level stores that ignore env overrides), confirm the launch path propagates the isolation to child processes, and back up live data before the first run — a mismatched schema that drops unknown fields on write is data loss, not a no-op.
82
83
  - Sandbox, replay, or re-adjudication runs on production-derived config: enumerate every outbound channel the stage can reach — publish, upload, notify, external write — and disable or redirect each one before the run, proving each disarm fires as you would prove a path guard; a guard on the input or target path alone leaves egress armed. Fingerprint every external destination before the run and diff it after, so an escaped write is caught by the run rather than by a recipient.
@@ -151,6 +152,11 @@ produce a green with no evidence behind it:
151
152
  - **The suspiciously fast or empty run.** When a check goes green unexpectedly quickly, or reports
152
153
  nothing at all, dump what it actually ran over before believing it. A harness that crashed early
153
154
  and one that found nothing produce the same exit code.
155
+ - **The verdict that ran before its checks.** If a runner computes or prints its overall verdict
156
+ before all checks finish, later failures cannot change it. Accumulate one failure count across
157
+ every check, then compute the verdict and exit status once, after the last check. Read a
158
+ multi-check run from that final count and an exit status verified to derive from it — never from
159
+ `tail` or collapsed last lines, which hide failures printed earlier.
154
160
  - **The control that failed by crashing.** A negative control is evidence only when it fails
155
161
  through the assertion it names: a traceback and a caught violation share an exit code, and an
156
162
  early crash can pre-empt every control after it. Treat each traceback in a control run as a
@@ -173,7 +179,10 @@ produce a green with no evidence behind it:
173
179
  - **The mutant that never ran.** A mutation verdict counts only if the mutant is valid: it
174
180
  compiled, sits on a path the exercised test traverses, and changes the guarded behavior, not
175
181
  healed downstream or coinciding with a default. The runner must report build failure,
176
- unreachable, and equivalent distinctly from KILLED and SURVIVED, and halt on a moved anchor.
182
+ unreachable, equivalent, and timed out distinctly from KILLED and SURVIVED, and halt on a moved
183
+ anchor. A timeout is not a kill: it moves with machine load, so set the limit well above the
184
+ unmutated suite's runtime and accept a verdict only when repeated runs agree on which mutants
185
+ timed out.
177
186
  More tests red than the mutation should touch indicts it; classify a survivor (rebuild,
178
187
  discard, genuine gap) before writing a test. **Equivalent is a verdict about the probe as
179
188
  much as the mutant**: a probe that is dead or returns a constant reports every mutant as
@@ -18,7 +18,29 @@ GUIDE = "guides/tooling-gotchas.md"
18
18
  # Both hosts receive the same command payload and additionalContext response.
19
19
  # The guide remains available when native hooks are disabled, untrusted, or do
20
20
  # not cover a tool path. --self-test checks every anchor against both guide trees.
21
+ _SECRET_FILE = (r"(?:\.env(?:\.\w+)?|\.netrc|\.pgpass|\.npmrc|\.pypirc|credentials(?:\.json)?"
22
+ r"|auth\.json|secrets?\.(?:ya?ml|json|toml))\b")
21
23
  RULES = [
24
+ # First because it is the one reminder whose miss is not re-work but exposure. Each
25
+ # branch is a command whose ordinary output IS the value: a keychain read with -w/-g,
26
+ # printenv or a bare env, a token-printing CLI, an echo of a credential-named variable
27
+ # (`\w*TOKEN\b` leaves $MAX_TOKENS alone), and a display or match over a secret-bearing
28
+ # file. A grep with -l/-L/-c/-q prints names, counts or nothing, so it stays quiet.
29
+ ("credential-print",
30
+ re.compile(r"(?i)\bsecurity\s+find-(?:generic|internet)-password\b[^|;&]*\s-[a-z]*[wg]\b"
31
+ r"|\bprintenv\b(?!\s+(?:PATH|HOME|SHELL|USER|PWD|LANG|TERM|TMPDIR)\b)"
32
+ r"|(?:^|[|;&]\s*)env\s*(?:$|[|;&])"
33
+ r"|\bgcloud\s+auth\s+(?:application-default\s+)?print-(?:access|identity)-token\b"
34
+ r"|\bgh\s+auth\s+token\b"
35
+ r"|\bkubectl\s+get\s+secrets?\b[^|;&]*-o\s*=?\s*(?:ya?ml|json|jsonpath)"
36
+ r"|\b(?:echo|printf)\b[^|;&]*\$\{?\w*(?:TOKEN|SECRET|PASSWORD|PASSWD|API_?KEY"
37
+ r"|PRIVATE_?KEY|ACCESS_?KEY|CREDENTIALS?)\b"
38
+ r"|\b(?:cat|head|tail|less|more|bat)\b[^|;&]*" + _SECRET_FILE +
39
+ r"|\b(?:grep|rg)\b(?![^|;&]*\s-[a-zA-Z]*[lLcq])[^|;&]*" + _SECRET_FILE),
40
+ "Checking where a credential lives or that it is set must not print it — test "
41
+ "existence, length, or shape without emitting secret characters; a value that "
42
+ f"reached the transcript is exposed, rotate it ({GUIDE}).",
43
+ "Locating a credential must not print it"),
22
44
  ("reserved-shell-names",
23
45
  re.compile(r"\b(UID|EUID|GID|PPID)="),
24
46
  "Assigning reserved shell names (UID/EUID/GID/PPID) can invoke the bound "
@@ -280,6 +302,7 @@ def self_test() -> int:
280
302
  # matches() never reaches the caller either — both leave the rule inert while this file
281
303
  # goes on reporting that all of them fire.
282
304
  FIXTURES = {
305
+ "credential-print": "cat .env",
283
306
  "reserved-shell-names": "UID=0 echo hi",
284
307
  "git-diff-two-dot": "git diff main..HEAD",
285
308
  "git-pull-dirty": "git pull origin main",
@@ -417,6 +440,24 @@ def self_test() -> int:
417
440
  if "grep-binary-heuristic" in [h for h, _ in matches("(grep -a needle payload)", limit=None)]:
418
441
  problems.append("grep-binary-heuristic: fired although the subshell-grouped "
419
442
  "grep carries its own text-mode flag")
443
+ # Every branch of credential-print is a separate way the value reaches the transcript,
444
+ # so each gets its own firing case — a regex that loses one alternative still fires on
445
+ # `cat .env` and would pass the fixture above. The quiet cases are the look-alikes the
446
+ # rule must leave alone: existence and length checks, name-only and count-only greps,
447
+ # a variable named for LLM token counts, a keychain lookup without -w, and a printenv
448
+ # of a non-secret variable.
449
+ for cmd in ("printenv", "env | grep KEY", "echo $GITHUB_TOKEN", 'echo "${OPENAI_API_KEY}"',
450
+ "printf '%s' $DB_PASSWORD", "security find-generic-password -s svc -w",
451
+ "gcloud auth print-access-token", "gh auth token", "grep API_KEY .env",
452
+ "kubectl get secret db -o yaml", "head -3 credentials.json", "tail ~/.netrc"):
453
+ if "credential-print" not in [h for h, _ in matches(cmd, limit=None)]:
454
+ problems.append(f"credential-print: does not fire on a value-printing command ({cmd!r})")
455
+ for cmd in ('[ -n "${GH_TOKEN:-}" ]', "wc -c < .env", "grep -c API_KEY .env", "grep -l token .env",
456
+ "echo $MAX_TOKENS", "security find-generic-password -s svc", "gh auth status",
457
+ "gcloud auth list", "cat README.md", "env FOO=1 python3 x.py", "printenv PATH",
458
+ "grep -rn token src/"):
459
+ if "credential-print" in [h for h, _ in matches(cmd, limit=None)]:
460
+ problems.append(f"credential-print: fired on a command that prints no secret ({cmd!r})")
420
461
  for envcmd in ("LC_ALL=C grep needle payload", "cat payload | LC_ALL=C grep needle"):
421
462
  if "grep-binary-heuristic" not in [h for h, _ in matches(envcmd, limit=None)]:
422
463
  problems.append(f"grep-binary-heuristic: an env-assignment prefix hid the "
@@ -5,8 +5,8 @@ description: Write or overhaul a repository's AGENTS.md (with CLAUDE.md as a one
5
5
 
6
6
  # Repo charter
7
7
 
8
- A repository's AGENTS.md is the repo's own layer of agent instruction — what a global
9
- corpus cannot know because it is true only here. This skill produces that layer for a real
8
+ A repository's AGENTS.md is the repo's own layer of agent instruction — what global
9
+ instructions cannot supply because it is true only here. This skill produces that layer for a real
10
10
  repository, and it produces it *from the repository*: the substantive half is reading the
11
11
  invariants, the gates and the traps out of the code, and no template can do that part.
12
12
 
@@ -40,7 +40,7 @@ thick; everything else stays thin or absent.
40
40
  | Application / product | project invariants, pitfall warnings, co-change duties |
41
41
  | CLI / single-author tool | repo orientation, project invariants, task procedure |
42
42
  | Monorepo / platform | context routing, task procedure |
43
- | Corpus / payload the repo publishes elsewhere (the code is delivery, the content is the product) | hard boundaries, co-change duties, completion gates |
43
+ | Instructions / payload the repo publishes elsewhere (the code is delivery, the content is the product) | hard boundaries, co-change duties, completion gates |
44
44
  | Content / docs | repo orientation only, minimal |
45
45
 
46
46
  A repo may take two rows; take the union and note which row explains each thick category.