catalogify 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. catalogify-0.5.0/LICENSE +21 -0
  2. catalogify-0.5.0/PKG-INFO +322 -0
  3. catalogify-0.5.0/README.md +302 -0
  4. catalogify-0.5.0/pyproject.toml +42 -0
  5. catalogify-0.5.0/setup.cfg +4 -0
  6. catalogify-0.5.0/src/catalogify/__init__.py +11 -0
  7. catalogify-0.5.0/src/catalogify/_scripts/__init__.py +0 -0
  8. catalogify-0.5.0/src/catalogify/_scripts/okf-history.sh +157 -0
  9. catalogify-0.5.0/src/catalogify/_scripts/okf-inventory.sh +300 -0
  10. catalogify-0.5.0/src/catalogify/_scripts/validate_okf.py +356 -0
  11. catalogify-0.5.0/src/catalogify/_skill_assets/SKILL.md +135 -0
  12. catalogify-0.5.0/src/catalogify/_skill_assets/__init__.py +0 -0
  13. catalogify-0.5.0/src/catalogify/_skill_assets/okf-config.template.yml +67 -0
  14. catalogify-0.5.0/src/catalogify/_skill_assets/references/clarify.md +102 -0
  15. catalogify-0.5.0/src/catalogify/_skill_assets/references/generate.md +384 -0
  16. catalogify-0.5.0/src/catalogify/_skill_assets/references/update.md +133 -0
  17. catalogify-0.5.0/src/catalogify/_skill_assets/references/validate.md +44 -0
  18. catalogify-0.5.0/src/catalogify/cli.py +71 -0
  19. catalogify-0.5.0/src/catalogify/installer.py +175 -0
  20. catalogify-0.5.0/src/catalogify/runner.py +78 -0
  21. catalogify-0.5.0/src/catalogify.egg-info/PKG-INFO +322 -0
  22. catalogify-0.5.0/src/catalogify.egg-info/SOURCES.txt +24 -0
  23. catalogify-0.5.0/src/catalogify.egg-info/dependency_links.txt +1 -0
  24. catalogify-0.5.0/src/catalogify.egg-info/entry_points.txt +5 -0
  25. catalogify-0.5.0/src/catalogify.egg-info/requires.txt +1 -0
  26. catalogify-0.5.0/src/catalogify.egg-info/top_level.txt +1 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Alex Punnen
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,322 @@
1
+ Metadata-Version: 2.4
2
+ Name: catalogify
3
+ Version: 0.5.0
4
+ Summary: Turn a repository into a knowledge catalog an AI agent can afford to read: Open Knowledge Format concepts mined from code and git history.
5
+ Author-email: Alex Punnen <alexcpn@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/alexcpn/catalogify
8
+ Project-URL: Repository, https://github.com/alexcpn/catalogify
9
+ Project-URL: Issues, https://github.com/alexcpn/catalogify/issues
10
+ Keywords: ai,agents,knowledge-graph,documentation,okf,git,claude,llm
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Software Development :: Documentation
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: PyYAML>=6.0
19
+ Dynamic: license-file
20
+
21
+ # catalogify
22
+
23
+ **Turn a repository into a knowledge catalog your AI agent can afford to read.**
24
+
25
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
26
+
27
+ `catalogify` generates an [Open Knowledge Format (OKF v0.1)](https://github.com/GoogleCloudPlatform/knowledge-catalog/blob/main/okf/SPEC.md)
28
+ bundle: a directory of cross-linked markdown concepts with YAML frontmatter
29
+ describing a codebase's services, modules, APIs, data models and operations.
30
+ Where git is available it mines the history for the reasoning behind the code,
31
+ and whatever it cannot establish it parks as an open question rather than
32
+ inventing an answer.
33
+
34
+ - **Monorepo or single repo.** One `Service` concept per deployable unit, scoped by folder — a repo with hundreds of services produces the same directory layout as a single-service repo, just wider.
35
+ - **Cheap on large codebases.** Scanning 25,917 files and 500,022 lines of Kubernetes takes 2.1 seconds. A service entry is ~676 tokens whether the service is 10k lines or 100k.
36
+ - **Git optional.** History mining and incremental updates need it; everything else does not.
37
+
38
+ It installs as a portable **Agent Skill**, so it works in Claude Code, Cursor,
39
+ OpenAI Codex, and anything else that reads the open `.agents/skills` standard.
40
+
41
+ ```bash
42
+ uv tool install catalogify
43
+ catalogify install
44
+ ```
45
+
46
+ Then ask, in plain language:
47
+
48
+ ```
49
+ "generate a knowledge catalog for this repo"
50
+ ```
51
+
52
+ ## Why it exists
53
+
54
+ Giving an agent "context on the codebase" is two problems.
55
+
56
+ **Routing** — *which of our 90 services does this spec touch?* Wide and
57
+ shallow. You need a little about everything.
58
+
59
+ **Reaching** — *inside that service, what changes?* Narrow and deep. You need
60
+ everything about a little.
61
+
62
+ A code graph such as [Graphify](https://github.com/Graphify-Labs/graphify) —
63
+ which parses your code with tree-sitter and builds a queryable graph of every
64
+ symbol and call edge — is excellent at reaching and will beat prose every
65
+ time. Ask it "what breaks if I change this function" and it answers precisely.
66
+ `catalogify` targets routing instead, where the winning property is being
67
+ small enough that an agent can read the whole estate in one call.
68
+
69
+ The two compose well: route with catalogify to pick the services, then run
70
+ Graphify inside the one you picked. They are not alternatives.
71
+
72
+ Measured on `pkg/kubelet` from `kubernetes/kubernetes` (108,648 lines of Go),
73
+ with Graphify run over the same directory:
74
+
75
+ | Artifact | Size | Tokens |
76
+ | --- | ---: | ---: |
77
+ | Graphify `graph.json` | 14.5 MB | 3,813,486 |
78
+ | Graphify `wiki/` (446 articles) | 1,004 KB | 256,968 |
79
+ | catalogify bundle (9 concepts) | 21.9 KB | 5,596 |
80
+ | **catalogify service entry** | **2.6 KB** | **676** |
81
+
82
+ At 676 tokens per service, a 90-service catalog is roughly **61,000 tokens**
83
+ and fits in one call alongside the specification. See [Benchmark](#benchmark)
84
+ to reproduce these numbers.
85
+
86
+ ## What it runs on
87
+
88
+ **Monorepos and single repos alike.** Concepts are scoped by folder, so a
89
+ repository holding hundreds of services gets one `Service` concept per
90
+ deployable unit and its own `modules/`, `apis/` and `data/` concepts
91
+ underneath — the same directory layout a single-service repo produces, just
92
+ wider. Point it at the whole tree or at one subdirectory.
93
+
94
+ **Large codebases stay cheap**, because the catalog describes the repo rather
95
+ than reproducing it. Scanning all of `kubernetes/kubernetes` — **25,917 files
96
+ and 500,022 lines of Go** — takes **2.1 seconds** and yields a 56 KB inventory
97
+ (~14,000 tokens) for the agent to plan from. Cost scales with the number of
98
+ things worth naming, not with lines of code: a service entry is ~676 tokens
99
+ whether the service is 10k lines or 100k.
100
+
101
+ **With or without git.** History mining is a bonus, not a requirement:
102
+
103
+ | | With git | Without git |
104
+ | --- | --- | --- |
105
+ | Inventory, concepts, indexes, validation | yes | yes |
106
+ | Churn ranking, the "why" from reverts and hotfixes | yes | — |
107
+ | Commit citations, `resource:` URLs from the remote | yes | — |
108
+ | Incremental `update` | yes | — (re-run `generate`) |
109
+ | Conformant bundle | yes | yes |
110
+
111
+ Outside a repository the inventory reports `git.is_git_repo: false`, `history`
112
+ prints a notice and exits cleanly, timestamps come from file modification
113
+ times, and `log.md` records ``Commit: `none` ``, which the validator accepts.
114
+ The agent is told to raise more `open_questions` in that case, since without
115
+ history the reasoning behind the code can only come from you.
116
+
117
+ ## What a concept looks like
118
+
119
+ ```markdown
120
+ ---
121
+ type: Module
122
+ title: Container Manager (cm)
123
+ description: Owns cgroup hierarchy, CPU/memory/device allocation, and the
124
+ on-disk checkpoints that let those allocations survive a kubelet restart.
125
+ tags: [cgroups, cpumanager, checkpoint, qos]
126
+ source_files: [pkg/kubelet/cm]
127
+ open_questions:
128
+ - "After the V2→V3 migration fix was reverted, is the V3→V2 hybrid-state
129
+ hazard still live, or was it addressed another way?"
130
+ ---
131
+
132
+ # Gotchas
133
+
134
+ Checkpoint format changes are the highest-risk edit in this module, and the
135
+ hazard is the fallback path. When a V3 checkpoint has an invalid checksum,
136
+ restore falls back to V2, but the V3 fields already read stay in the struct,
137
+ producing a hybrid of V2 and V3 data (`83f1cae9656`, reverted by
138
+ `76100602564`).
139
+ ```
140
+
141
+ That gotcha is not in a comment, a docstring, or an ADR. It was recovered from
142
+ the commit log by following a revert back to the commit it reverted.
143
+
144
+ `source_files` maps the concept to code, which is what makes incremental
145
+ updates possible. `open_questions` is where uncertainty goes instead of into
146
+ prose. Both are producer extension fields permitted by OKF §4.1.
147
+
148
+ ## Why mine history at all
149
+
150
+ The gotcha in that example exists in no comment, no docstring, and no design
151
+ document — in Kubernetes, a project with KEPs, a design-proposals archive, and
152
+ reviewers who demand rationale. It survived only as a revert.
153
+
154
+ That is not an oversight, it is the normal condition. Michael Polanyi called it
155
+ tacit knowledge in 1966: *"we know more than we can tell."* Peter Naur applied
156
+ it to software in [*Programming as Theory Building*](https://pages.cs.wisc.edu/~remzi/Naur.pdf)
157
+ (1985), arguing that the real product of programming is the **theory** of the
158
+ system held in the developers' heads, and that program text and documentation
159
+ are insufficient carriers of it. A program whose original team has dispersed is,
160
+ in his terms, dead — and a new team patching it produces characteristically
161
+ wrong fixes that erode the system's conceptual integrity.
162
+
163
+ That is precisely the position an AI agent is in on first contact with your
164
+ repository. It arrives after the team has dispersed, holding the artefacts and
165
+ none of the theory.
166
+
167
+ Nor does writing a specification escape it. Fred Brooks, in *No Silver Bullet*:
168
+ *"the complexity of software is an essential property, not an accidental one,"*
169
+ so *"descriptions of a software entity that abstract away its complexity often
170
+ abstract away its essence."* That ceiling applies whether a human or a model
171
+ wrote the spec.
172
+
173
+ Commit history is a narrow exception. Nobody writes a revert as documentation;
174
+ they write it because production broke, leaving a dated, attributed, immutable
175
+ record of a place where the theory and the code disagreed. `catalogify history`
176
+ goes looking for exactly those.
177
+
178
+ This recovers fragments, not the theory. For everything still missing, the
179
+ generator raises an `open_question` and the clarify workflow asks a human while
180
+ there is still a human to ask.
181
+
182
+ ## The four workflows
183
+
184
+ Ask in plain language and the skill picks one:
185
+
186
+ | Workflow | What it does |
187
+ | -------- | ------------ |
188
+ | **generate** | Inventory scan, git-history mining, concept plan, concept documents, `index.md` files, `log.md`, validation. |
189
+ | **update** | Diffs since the last logged commit; refreshes only stale concepts, deprecates orphans, adds new ones, preserves human curation. |
190
+ | **clarify** | Asks you about the `open_questions` the other workflows parked, then folds the answers in as cited, curation-protected knowledge. |
191
+ | **validate** | OKF §9 conformance check plus a quality spot-check. |
192
+
193
+ ## Commands
194
+
195
+ The deterministic work is three subcommands you can run yourself, with or
196
+ without an agent. Each accepts `--help`.
197
+
198
+ ```bash
199
+ catalogify inventory # repo facts + git churn, as JSON
200
+ catalogify history pkg/foo --limit 5 # the reverts and hotfixes behind a path
201
+ catalogify validate knowledge/ # OKF v0.1 §9 conformance
202
+ catalogify install [--list] [--uninstall] # manage the agent skill
203
+ ```
204
+
205
+ - **`inventory`** writes JSON: file tree, languages, entry points, dependency manifests, API definitions, schemas, CI/CD, docs, ADRs, plus per-file commit churn. On the full Kubernetes tree (500k lines, 25,917 files) it takes 2.1 seconds and produces 56 KB.
206
+ - **`history`** returns the creation commit, recent subjects, and the revert / hotfix / risk-flagged commits where invariants hide. Diff-free by default so historical secrets do not leak; `--patch` opts in.
207
+ - **`validate`** enforces OKF §9: four error classes, nine warning classes.
208
+
209
+ `okf-inventory`, `okf-history` and `okf-validate` remain as aliases from the
210
+ package's previous life as `okf_skill`.
211
+
212
+ ## Requirements
213
+
214
+ - **Python 3.9+**
215
+ - **`bash`** — present on Linux and macOS; on Windows the wrapper finds the `bash.exe` that ships with [Git for Windows](https://git-scm.com/download/win).
216
+ - **`git`** — *optional*. Used for churn ranking, history mining and incremental updates. Everything else works without it; see [What it runs on](#what-it-runs-on).
217
+
218
+ ## Install
219
+
220
+ ```bash
221
+ uv tool install catalogify # or: pip install catalogify
222
+ catalogify install
223
+ ```
224
+
225
+ Restart your agent afterwards so it picks up the skill. `catalogify install`
226
+ copies it into every agent's user-global skills directory
227
+ (`~/.claude/skills`, `~/.cursor/skills`, `~/.codex/skills`,
228
+ `~/.agents/skills`). Use `--agents claude,cursor` to target specific ones and
229
+ `--scope project` to install into `./.<agent>/skills` instead.
230
+
231
+ The bundled `install.sh` / `install.ps1` do both steps, install `uv` if it is
232
+ missing, and fall back to `pip install --user`.
233
+
234
+ ## Usage
235
+
236
+ ```bash
237
+ agent "generate a knowledge catalog for this repo"
238
+ agent "refresh the catalog, the code has moved on"
239
+ agent "resolve the open questions in the catalog"
240
+ agent "validate the catalog"
241
+ ```
242
+
243
+ Output lands in `knowledge/` (configurable), ready to commit next to the code:
244
+
245
+ ```
246
+ knowledge/
247
+ ├── index.md # okf_version: "0.1" + directory of everything
248
+ ├── log.md # dated history, each block records a commit SHA
249
+ ├── architecture/
250
+ │ └── overview.md # type: Reference — the "start here" concept
251
+ ├── services/… # type: Service
252
+ ├── modules/… # type: Module
253
+ ├── apis/… # type: API Endpoint / API Resource
254
+ ├── data/… # type: Data Model / Database Table
255
+ └── operations/… # type: Pipeline / Configuration / Playbook
256
+ ```
257
+
258
+ On a monorepo, start with `granularity: coarse` and raise
259
+ `OKF_INVENTORY_CAP` above its default of 150. Raw churn skews toward generated
260
+ files and build config, so invest in the `exclude` list.
261
+
262
+ ## Configuration
263
+
264
+ Everything works with no config. To change the bundle directory, resource URI
265
+ base, excludes, type mappings, layout, granularity, or the clarify question
266
+ budget, copy the annotated template into your repo root:
267
+
268
+ ```bash
269
+ cp ~/.claude/skills/catalogify/okf-config.template.yml .okf-config.yml
270
+ ```
271
+
272
+ `catalogify inventory --config .okf-config.yml` and `catalogify validate <dir>
273
+ --config .okf-config.yml` honor its `exclude` list; the agent passes it through
274
+ automatically once the file exists.
275
+
276
+ ## Safety properties
277
+
278
+ - **Never guesses.** Unverifiable facts become `open_questions`, resolved by the clarify workflow and marked with `<!-- clarified: ... -->` sentinels that later updates will not overwrite. Your answer outranks the machine's inference permanently.
279
+ - **Never deletes curation.** Removed code marks a concept `status: deprecated` rather than deleting it. Human prose survives every refresh.
280
+ - **Never emits secrets.** Config values are described by shape, never value, including from history. The validator flags anything that slips through (W5).
281
+ - **Never touches source code.** All writes stay inside the bundle directory.
282
+
283
+ ## Benchmark
284
+
285
+ To reproduce the table above:
286
+
287
+ ```bash
288
+ git clone --filter=blob:none --no-tags \
289
+ https://github.com/kubernetes/kubernetes.git k8s
290
+
291
+ # Graphify, for comparison
292
+ pip install graphifyy
293
+ graphify update k8s/pkg/kubelet
294
+ cd k8s/pkg/kubelet && graphify export wiki
295
+ wc -c graphify-out/graph.json # 15,253,944
296
+ cat graphify-out/wiki/*.md | wc -c # 1,027,874
297
+
298
+ # catalogify
299
+ cd ../.. # back to the k8s repo root
300
+ catalogify inventory # 2.1s, 56 KB of JSON
301
+ catalogify history pkg/kubelet/cm --limit 3
302
+
303
+ # then ask your agent to generate the catalog, and check it:
304
+ catalogify validate knowledge/
305
+ ```
306
+
307
+ Token counts are bytes ÷ 4. Measured against `kubernetes/kubernetes` at commit
308
+ `d5ccf7968e5`. The structural graph was built AST-only (no API key), so its
309
+ wiki lacks LLM community labels.
310
+
311
+ ## Uninstall
312
+
313
+ ```bash
314
+ catalogify install --uninstall # remove from every agent's skills dir
315
+ uv tool uninstall catalogify
316
+ ```
317
+
318
+ Run `catalogify install --list` first to see where it is installed.
319
+
320
+ ## License
321
+
322
+ MIT
@@ -0,0 +1,302 @@
1
+ # catalogify
2
+
3
+ **Turn a repository into a knowledge catalog your AI agent can afford to read.**
4
+
5
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
6
+
7
+ `catalogify` generates an [Open Knowledge Format (OKF v0.1)](https://github.com/GoogleCloudPlatform/knowledge-catalog/blob/main/okf/SPEC.md)
8
+ bundle: a directory of cross-linked markdown concepts with YAML frontmatter
9
+ describing a codebase's services, modules, APIs, data models and operations.
10
+ Where git is available it mines the history for the reasoning behind the code,
11
+ and whatever it cannot establish it parks as an open question rather than
12
+ inventing an answer.
13
+
14
+ - **Monorepo or single repo.** One `Service` concept per deployable unit, scoped by folder — a repo with hundreds of services produces the same directory layout as a single-service repo, just wider.
15
+ - **Cheap on large codebases.** Scanning 25,917 files and 500,022 lines of Kubernetes takes 2.1 seconds. A service entry is ~676 tokens whether the service is 10k lines or 100k.
16
+ - **Git optional.** History mining and incremental updates need it; everything else does not.
17
+
18
+ It installs as a portable **Agent Skill**, so it works in Claude Code, Cursor,
19
+ OpenAI Codex, and anything else that reads the open `.agents/skills` standard.
20
+
21
+ ```bash
22
+ uv tool install catalogify
23
+ catalogify install
24
+ ```
25
+
26
+ Then ask, in plain language:
27
+
28
+ ```
29
+ "generate a knowledge catalog for this repo"
30
+ ```
31
+
32
+ ## Why it exists
33
+
34
+ Giving an agent "context on the codebase" is two problems.
35
+
36
+ **Routing** — *which of our 90 services does this spec touch?* Wide and
37
+ shallow. You need a little about everything.
38
+
39
+ **Reaching** — *inside that service, what changes?* Narrow and deep. You need
40
+ everything about a little.
41
+
42
+ A code graph such as [Graphify](https://github.com/Graphify-Labs/graphify) —
43
+ which parses your code with tree-sitter and builds a queryable graph of every
44
+ symbol and call edge — is excellent at reaching and will beat prose every
45
+ time. Ask it "what breaks if I change this function" and it answers precisely.
46
+ `catalogify` targets routing instead, where the winning property is being
47
+ small enough that an agent can read the whole estate in one call.
48
+
49
+ The two compose well: route with catalogify to pick the services, then run
50
+ Graphify inside the one you picked. They are not alternatives.
51
+
52
+ Measured on `pkg/kubelet` from `kubernetes/kubernetes` (108,648 lines of Go),
53
+ with Graphify run over the same directory:
54
+
55
+ | Artifact | Size | Tokens |
56
+ | --- | ---: | ---: |
57
+ | Graphify `graph.json` | 14.5 MB | 3,813,486 |
58
+ | Graphify `wiki/` (446 articles) | 1,004 KB | 256,968 |
59
+ | catalogify bundle (9 concepts) | 21.9 KB | 5,596 |
60
+ | **catalogify service entry** | **2.6 KB** | **676** |
61
+
62
+ At 676 tokens per service, a 90-service catalog is roughly **61,000 tokens**
63
+ and fits in one call alongside the specification. See [Benchmark](#benchmark)
64
+ to reproduce these numbers.
65
+
66
+ ## What it runs on
67
+
68
+ **Monorepos and single repos alike.** Concepts are scoped by folder, so a
69
+ repository holding hundreds of services gets one `Service` concept per
70
+ deployable unit and its own `modules/`, `apis/` and `data/` concepts
71
+ underneath — the same directory layout a single-service repo produces, just
72
+ wider. Point it at the whole tree or at one subdirectory.
73
+
74
+ **Large codebases stay cheap**, because the catalog describes the repo rather
75
+ than reproducing it. Scanning all of `kubernetes/kubernetes` — **25,917 files
76
+ and 500,022 lines of Go** — takes **2.1 seconds** and yields a 56 KB inventory
77
+ (~14,000 tokens) for the agent to plan from. Cost scales with the number of
78
+ things worth naming, not with lines of code: a service entry is ~676 tokens
79
+ whether the service is 10k lines or 100k.
80
+
81
+ **With or without git.** History mining is a bonus, not a requirement:
82
+
83
+ | | With git | Without git |
84
+ | --- | --- | --- |
85
+ | Inventory, concepts, indexes, validation | yes | yes |
86
+ | Churn ranking, the "why" from reverts and hotfixes | yes | — |
87
+ | Commit citations, `resource:` URLs from the remote | yes | — |
88
+ | Incremental `update` | yes | — (re-run `generate`) |
89
+ | Conformant bundle | yes | yes |
90
+
91
+ Outside a repository the inventory reports `git.is_git_repo: false`, `history`
92
+ prints a notice and exits cleanly, timestamps come from file modification
93
+ times, and `log.md` records ``Commit: `none` ``, which the validator accepts.
94
+ The agent is told to raise more `open_questions` in that case, since without
95
+ history the reasoning behind the code can only come from you.
96
+
97
+ ## What a concept looks like
98
+
99
+ ```markdown
100
+ ---
101
+ type: Module
102
+ title: Container Manager (cm)
103
+ description: Owns cgroup hierarchy, CPU/memory/device allocation, and the
104
+ on-disk checkpoints that let those allocations survive a kubelet restart.
105
+ tags: [cgroups, cpumanager, checkpoint, qos]
106
+ source_files: [pkg/kubelet/cm]
107
+ open_questions:
108
+ - "After the V2→V3 migration fix was reverted, is the V3→V2 hybrid-state
109
+ hazard still live, or was it addressed another way?"
110
+ ---
111
+
112
+ # Gotchas
113
+
114
+ Checkpoint format changes are the highest-risk edit in this module, and the
115
+ hazard is the fallback path. When a V3 checkpoint has an invalid checksum,
116
+ restore falls back to V2, but the V3 fields already read stay in the struct,
117
+ producing a hybrid of V2 and V3 data (`83f1cae9656`, reverted by
118
+ `76100602564`).
119
+ ```
120
+
121
+ That gotcha is not in a comment, a docstring, or an ADR. It was recovered from
122
+ the commit log by following a revert back to the commit it reverted.
123
+
124
+ `source_files` maps the concept to code, which is what makes incremental
125
+ updates possible. `open_questions` is where uncertainty goes instead of into
126
+ prose. Both are producer extension fields permitted by OKF §4.1.
127
+
128
+ ## Why mine history at all
129
+
130
+ The gotcha in that example exists in no comment, no docstring, and no design
131
+ document — in Kubernetes, a project with KEPs, a design-proposals archive, and
132
+ reviewers who demand rationale. It survived only as a revert.
133
+
134
+ That is not an oversight, it is the normal condition. Michael Polanyi called it
135
+ tacit knowledge in 1966: *"we know more than we can tell."* Peter Naur applied
136
+ it to software in [*Programming as Theory Building*](https://pages.cs.wisc.edu/~remzi/Naur.pdf)
137
+ (1985), arguing that the real product of programming is the **theory** of the
138
+ system held in the developers' heads, and that program text and documentation
139
+ are insufficient carriers of it. A program whose original team has dispersed is,
140
+ in his terms, dead — and a new team patching it produces characteristically
141
+ wrong fixes that erode the system's conceptual integrity.
142
+
143
+ That is precisely the position an AI agent is in on first contact with your
144
+ repository. It arrives after the team has dispersed, holding the artefacts and
145
+ none of the theory.
146
+
147
+ Nor does writing a specification escape it. Fred Brooks, in *No Silver Bullet*:
148
+ *"the complexity of software is an essential property, not an accidental one,"*
149
+ so *"descriptions of a software entity that abstract away its complexity often
150
+ abstract away its essence."* That ceiling applies whether a human or a model
151
+ wrote the spec.
152
+
153
+ Commit history is a narrow exception. Nobody writes a revert as documentation;
154
+ they write it because production broke, leaving a dated, attributed, immutable
155
+ record of a place where the theory and the code disagreed. `catalogify history`
156
+ goes looking for exactly those.
157
+
158
+ This recovers fragments, not the theory. For everything still missing, the
159
+ generator raises an `open_question` and the clarify workflow asks a human while
160
+ there is still a human to ask.
161
+
162
+ ## The four workflows
163
+
164
+ Ask in plain language and the skill picks one:
165
+
166
+ | Workflow | What it does |
167
+ | -------- | ------------ |
168
+ | **generate** | Inventory scan, git-history mining, concept plan, concept documents, `index.md` files, `log.md`, validation. |
169
+ | **update** | Diffs since the last logged commit; refreshes only stale concepts, deprecates orphans, adds new ones, preserves human curation. |
170
+ | **clarify** | Asks you about the `open_questions` the other workflows parked, then folds the answers in as cited, curation-protected knowledge. |
171
+ | **validate** | OKF §9 conformance check plus a quality spot-check. |
172
+
173
+ ## Commands
174
+
175
+ The deterministic work is three subcommands you can run yourself, with or
176
+ without an agent. Each accepts `--help`.
177
+
178
+ ```bash
179
+ catalogify inventory # repo facts + git churn, as JSON
180
+ catalogify history pkg/foo --limit 5 # the reverts and hotfixes behind a path
181
+ catalogify validate knowledge/ # OKF v0.1 §9 conformance
182
+ catalogify install [--list] [--uninstall] # manage the agent skill
183
+ ```
184
+
185
+ - **`inventory`** writes JSON: file tree, languages, entry points, dependency manifests, API definitions, schemas, CI/CD, docs, ADRs, plus per-file commit churn. On the full Kubernetes tree (500k lines, 25,917 files) it takes 2.1 seconds and produces 56 KB.
186
+ - **`history`** returns the creation commit, recent subjects, and the revert / hotfix / risk-flagged commits where invariants hide. Diff-free by default so historical secrets do not leak; `--patch` opts in.
187
+ - **`validate`** enforces OKF §9: four error classes, nine warning classes.
188
+
189
+ `okf-inventory`, `okf-history` and `okf-validate` remain as aliases from the
190
+ package's previous life as `okf_skill`.
191
+
192
+ ## Requirements
193
+
194
+ - **Python 3.9+**
195
+ - **`bash`** — present on Linux and macOS; on Windows the wrapper finds the `bash.exe` that ships with [Git for Windows](https://git-scm.com/download/win).
196
+ - **`git`** — *optional*. Used for churn ranking, history mining and incremental updates. Everything else works without it; see [What it runs on](#what-it-runs-on).
197
+
198
+ ## Install
199
+
200
+ ```bash
201
+ uv tool install catalogify # or: pip install catalogify
202
+ catalogify install
203
+ ```
204
+
205
+ Restart your agent afterwards so it picks up the skill. `catalogify install`
206
+ copies it into every agent's user-global skills directory
207
+ (`~/.claude/skills`, `~/.cursor/skills`, `~/.codex/skills`,
208
+ `~/.agents/skills`). Use `--agents claude,cursor` to target specific ones and
209
+ `--scope project` to install into `./.<agent>/skills` instead.
210
+
211
+ The bundled `install.sh` / `install.ps1` do both steps, install `uv` if it is
212
+ missing, and fall back to `pip install --user`.
213
+
214
+ ## Usage
215
+
216
+ ```bash
217
+ agent "generate a knowledge catalog for this repo"
218
+ agent "refresh the catalog, the code has moved on"
219
+ agent "resolve the open questions in the catalog"
220
+ agent "validate the catalog"
221
+ ```
222
+
223
+ Output lands in `knowledge/` (configurable), ready to commit next to the code:
224
+
225
+ ```
226
+ knowledge/
227
+ ├── index.md # okf_version: "0.1" + directory of everything
228
+ ├── log.md # dated history, each block records a commit SHA
229
+ ├── architecture/
230
+ │ └── overview.md # type: Reference — the "start here" concept
231
+ ├── services/… # type: Service
232
+ ├── modules/… # type: Module
233
+ ├── apis/… # type: API Endpoint / API Resource
234
+ ├── data/… # type: Data Model / Database Table
235
+ └── operations/… # type: Pipeline / Configuration / Playbook
236
+ ```
237
+
238
+ On a monorepo, start with `granularity: coarse` and raise
239
+ `OKF_INVENTORY_CAP` above its default of 150. Raw churn skews toward generated
240
+ files and build config, so invest in the `exclude` list.
241
+
242
+ ## Configuration
243
+
244
+ Everything works with no config. To change the bundle directory, resource URI
245
+ base, excludes, type mappings, layout, granularity, or the clarify question
246
+ budget, copy the annotated template into your repo root:
247
+
248
+ ```bash
249
+ cp ~/.claude/skills/catalogify/okf-config.template.yml .okf-config.yml
250
+ ```
251
+
252
+ `catalogify inventory --config .okf-config.yml` and `catalogify validate <dir>
253
+ --config .okf-config.yml` honor its `exclude` list; the agent passes it through
254
+ automatically once the file exists.
255
+
256
+ ## Safety properties
257
+
258
+ - **Never guesses.** Unverifiable facts become `open_questions`, resolved by the clarify workflow and marked with `<!-- clarified: ... -->` sentinels that later updates will not overwrite. Your answer outranks the machine's inference permanently.
259
+ - **Never deletes curation.** Removed code marks a concept `status: deprecated` rather than deleting it. Human prose survives every refresh.
260
+ - **Never emits secrets.** Config values are described by shape, never value, including from history. The validator flags anything that slips through (W5).
261
+ - **Never touches source code.** All writes stay inside the bundle directory.
262
+
263
+ ## Benchmark
264
+
265
+ To reproduce the table above:
266
+
267
+ ```bash
268
+ git clone --filter=blob:none --no-tags \
269
+ https://github.com/kubernetes/kubernetes.git k8s
270
+
271
+ # Graphify, for comparison
272
+ pip install graphifyy
273
+ graphify update k8s/pkg/kubelet
274
+ cd k8s/pkg/kubelet && graphify export wiki
275
+ wc -c graphify-out/graph.json # 15,253,944
276
+ cat graphify-out/wiki/*.md | wc -c # 1,027,874
277
+
278
+ # catalogify
279
+ cd ../.. # back to the k8s repo root
280
+ catalogify inventory # 2.1s, 56 KB of JSON
281
+ catalogify history pkg/kubelet/cm --limit 3
282
+
283
+ # then ask your agent to generate the catalog, and check it:
284
+ catalogify validate knowledge/
285
+ ```
286
+
287
+ Token counts are bytes ÷ 4. Measured against `kubernetes/kubernetes` at commit
288
+ `d5ccf7968e5`. The structural graph was built AST-only (no API key), so its
289
+ wiki lacks LLM community labels.
290
+
291
+ ## Uninstall
292
+
293
+ ```bash
294
+ catalogify install --uninstall # remove from every agent's skills dir
295
+ uv tool uninstall catalogify
296
+ ```
297
+
298
+ Run `catalogify install --list` first to see where it is installed.
299
+
300
+ ## License
301
+
302
+ MIT
@@ -0,0 +1,42 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "catalogify"
7
+ version = "0.5.0"
8
+ description = "Turn a repository into a knowledge catalog an AI agent can afford to read: Open Knowledge Format concepts mined from code and git history."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ authors = [{ name = "Alex Punnen", email = "alexcpn@gmail.com" }]
13
+ keywords = ["ai", "agents", "knowledge-graph", "documentation", "okf", "git", "claude", "llm"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Developers",
17
+ "Programming Language :: Python :: 3",
18
+ "Topic :: Software Development :: Documentation",
19
+ ]
20
+ # PyYAML makes the validator's frontmatter parsing exact; it falls back to a
21
+ # lenient regex parser (and warns, W0) when it is missing.
22
+ dependencies = ["PyYAML>=6.0"]
23
+
24
+ [project.urls]
25
+ Homepage = "https://github.com/alexcpn/catalogify"
26
+ Repository = "https://github.com/alexcpn/catalogify"
27
+ Issues = "https://github.com/alexcpn/catalogify/issues"
28
+
29
+ [project.scripts]
30
+ catalogify = "catalogify.cli:main"
31
+ # Legacy aliases from the okf_skill package, kept so existing installs and
32
+ # older write-ups keep working. `catalogify <cmd>` is the supported form.
33
+ okf-inventory = "catalogify.runner:main_inventory"
34
+ okf-history = "catalogify.runner:main_history"
35
+ okf-validate = "catalogify.runner:main_validate"
36
+
37
+ [tool.setuptools.packages.find]
38
+ where = ["src"]
39
+
40
+ [tool.setuptools.package-data]
41
+ "catalogify._scripts" = ["*.sh", "*.py"]
42
+ "catalogify._skill_assets" = ["*.md", "*.yml", "references/*.md"]