awkno 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. awkno/__init__.py +25 -0
  2. awkno/_doctor.py +114 -0
  3. awkno/cli.py +497 -0
  4. awkno/corpus.py +216 -0
  5. awkno/generate.py +444 -0
  6. awkno/pages/agent-senses.json +13 -0
  7. awkno/pages/agent-vm.json +14 -0
  8. awkno/pages/aitherconnect.json +11 -0
  9. awkno/pages/aitherkvcache.json +11 -0
  10. awkno/pages/aitherzero.json +11 -0
  11. awkno/pages/awarena.json +14 -0
  12. awkno/pages/awask.json +15 -0
  13. awkno/pages/awbac.json +11 -0
  14. awkno/pages/awbrowse.json +13 -0
  15. awkno/pages/awdit.json +12 -0
  16. awkno/pages/awdk.json +16 -0
  17. awkno/pages/awevolve.json +15 -0
  18. awkno/pages/awfind.json +13 -0
  19. awkno/pages/awgit.json +12 -0
  20. awkno/pages/awgraph.json +12 -0
  21. awkno/pages/awiam.json +12 -0
  22. awkno/pages/awkit.json +13 -0
  23. awkno/pages/awkno.json +13 -0
  24. awkno/pages/awknowledge.json +12 -0
  25. awkno/pages/awm.json +13 -0
  26. awkno/pages/awmail.json +18 -0
  27. awkno/pages/awnboard.json +20 -0
  28. awkno/pages/awnest.json +19 -0
  29. awkno/pages/awnet.json +13 -0
  30. awkno/pages/awnix.json +20 -0
  31. awkno/pages/awnode.json +12 -0
  32. awkno/pages/awpack.json +13 -0
  33. awkno/pages/awpredict.json +11 -0
  34. awkno/pages/awprism.json +14 -0
  35. awkno/pages/awreason.json +14 -0
  36. awkno/pages/awrecover.json +13 -0
  37. awkno/pages/awrecurse.json +13 -0
  38. awkno/pages/awrelay.json +12 -0
  39. awkno/pages/awrepl.json +13 -0
  40. awkno/pages/awresearch.json +14 -0
  41. awkno/pages/awrun.json +13 -0
  42. awkno/pages/awseal.json +12 -0
  43. awkno/pages/awsh.json +13 -0
  44. awkno/pages/awshare.json +12 -0
  45. awkno/pages/awskills.json +12 -0
  46. awkno/pages/awsync.json +16 -0
  47. awkno/pages/awtunnel.json +12 -0
  48. awkno/pages/cited-research.json +14 -0
  49. awkno/pages/gobbonet-agentic.json +14 -0
  50. awkno/pages/guide-00.json +16 -0
  51. awkno/pages/guide-01.json +14 -0
  52. awkno/pages/guide-02.json +14 -0
  53. awkno/pages/guide-03.json +14 -0
  54. awkno/pages/guide-04.json +15 -0
  55. awkno/pages/guide-05.json +16 -0
  56. awkno/pages/guide-06.json +15 -0
  57. awkno/pages/guide-07.json +15 -0
  58. awkno/pages/guide-08.json +17 -0
  59. awkno/pages/guide-09.json +13 -0
  60. awkno/pages/guide.json +22 -0
  61. awkno/pages/law-01.json +11 -0
  62. awkno/pages/law-02.json +11 -0
  63. awkno/pages/law-03.json +11 -0
  64. awkno/pages/law-04.json +11 -0
  65. awkno/pages/law-05.json +11 -0
  66. awkno/pages/law-06.json +11 -0
  67. awkno/pages/law-07.json +11 -0
  68. awkno/pages/law-08.json +11 -0
  69. awkno/pages/law-09.json +11 -0
  70. awkno/pages/law-10.json +11 -0
  71. awkno/pages/law-11.json +11 -0
  72. awkno/pages/law-12.json +11 -0
  73. awkno/pages/law-13.json +11 -0
  74. awkno/pages/law-14.json +11 -0
  75. awkno/pages/law-15.json +11 -0
  76. awkno/pages/law-16.json +11 -0
  77. awkno/pages/law-17.json +11 -0
  78. awkno/pages/law-18.json +11 -0
  79. awkno/pages/law-19.json +11 -0
  80. awkno/pages/one-surface.json +15 -0
  81. awkno/pages/provenance.json +14 -0
  82. awkno/pages/shared-worktree.json +14 -0
  83. awkno/pages/the-front-door.json +15 -0
  84. awkno/pages/the-reasoning-loop.json +14 -0
  85. awkno/pages/who-what-did.json +13 -0
  86. awkno-0.2.0.dist-info/METADATA +321 -0
  87. awkno-0.2.0.dist-info/RECORD +91 -0
  88. awkno-0.2.0.dist-info/WHEEL +5 -0
  89. awkno-0.2.0.dist-info/entry_points.txt +2 -0
  90. awkno-0.2.0.dist-info/licenses/LICENSE +118 -0
  91. awkno-0.2.0.dist-info/top_level.txt +1 -0
awkno/corpus.py ADDED
@@ -0,0 +1,216 @@
1
+ """Load and query the offline corpus.
2
+
3
+ Pages are stored as JSON files under awkno/pages/. This module handles the
4
+ loading, searching, and formatting of those pages.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from dataclasses import dataclass
11
+ from difflib import SequenceMatcher
12
+ from pathlib import Path
13
+ from typing import Optional
14
+
15
+
16
+ class NotFoundError(Exception):
17
+ """Raised when a topic is not found."""
18
+
19
+ pass
20
+
21
+
22
+ @dataclass
23
+ class AwknoPage:
24
+ """A single man page."""
25
+
26
+ topic: str
27
+ category: str
28
+ synopsis: str
29
+ description: str
30
+ adopt: Optional[str] = None
31
+ status: Optional[str] = None
32
+ see_also: Optional[list[str]] = None
33
+ body: Optional[str] = None
34
+ # The law filename's own slug (`design-for-the-silence`). Carried rather
35
+ # than re-derived from the synopsis: the generator already computes it, and
36
+ # a slug inferred from a title silently stops matching the moment a law is
37
+ # retitled.
38
+ slug: Optional[str] = None
39
+
40
+ def render(self, plain: bool = False) -> str:
41
+ """Render the page as text.
42
+
43
+ Args:
44
+ plain: If True, no ANSI codes. If False, use formatting (when TTY).
45
+
46
+ Returns:
47
+ The formatted page text.
48
+ """
49
+ lines = []
50
+ lines.append("NAME")
51
+ lines.append(f" {self.topic} — {self.synopsis}")
52
+ lines.append("")
53
+
54
+ if self.adopt:
55
+ lines.append("ADOPT")
56
+ for line in self.adopt.split("\n"):
57
+ lines.append(f" {line}")
58
+ lines.append("")
59
+
60
+ if self.description:
61
+ lines.append("DESCRIPTION")
62
+ for line in self.description.split("\n"):
63
+ lines.append(f" {line}")
64
+ lines.append("")
65
+
66
+ if self.body:
67
+ lines.append("DETAILS")
68
+ for line in self.body.split("\n"):
69
+ lines.append(f" {line}")
70
+ lines.append("")
71
+
72
+ if self.status:
73
+ lines.append("STATUS")
74
+ lines.append(f" {self.status}")
75
+ lines.append("")
76
+
77
+ if self.see_also:
78
+ lines.append("SEE ALSO")
79
+ for item in self.see_also:
80
+ lines.append(f" {item}")
81
+
82
+ text = "\n".join(lines)
83
+
84
+ if not plain:
85
+ # Check if stdout is a TTY for formatting
86
+ import sys
87
+
88
+ if hasattr(sys.stdout, "isatty") and sys.stdout.isatty():
89
+ # Apply ANSI formatting
90
+ text = text.replace("NAME\n", "\033[1mNAME\033[0m\n")
91
+ text = text.replace("ADOPT\n", "\033[1mADOPT\033[0m\n")
92
+ text = text.replace("DESCRIPTION\n", "\033[1mDESCRIPTION\033[0m\n")
93
+ text = text.replace("DETAILS\n", "\033[1mDETAILS\033[0m\n")
94
+ text = text.replace("STATUS\n", "\033[1mSTATUS\033[0m\n")
95
+ text = text.replace("SEE ALSO\n", "\033[1mSEE ALSO\033[0m\n")
96
+
97
+ return text
98
+
99
+ def to_dict(self) -> dict:
100
+ """Convert to dict for JSON serialization."""
101
+ return {
102
+ "topic": self.topic,
103
+ "category": self.category,
104
+ "synopsis": self.synopsis,
105
+ "description": self.description,
106
+ "adopt": self.adopt,
107
+ "status": self.status,
108
+ "see_also": self.see_also,
109
+ "body": self.body,
110
+ "slug": self.slug,
111
+ }
112
+
113
+ @classmethod
114
+ def from_dict(cls, data: dict) -> AwknoPage:
115
+ """Create from dict (JSON deserialization)."""
116
+ return cls(**data)
117
+
118
+
119
+ class AwknoRegistry:
120
+ """Load and query all pages."""
121
+
122
+ def __init__(self):
123
+ """Load all pages from the corpus directory."""
124
+ self.pages: dict[str, AwknoPage] = {}
125
+ self._load_pages()
126
+
127
+ def _load_pages(self) -> None:
128
+ """Load all JSON files from awkno/pages/."""
129
+ pages_dir = Path(__file__).parent / "pages"
130
+
131
+ if not pages_dir.exists():
132
+ return
133
+
134
+ for json_file in pages_dir.glob("*.json"):
135
+ try:
136
+ with open(json_file) as f:
137
+ data = json.load(f)
138
+ page = AwknoPage.from_dict(data)
139
+ self.pages[page.topic.lower()] = page
140
+ except (json.JSONDecodeError, ValueError):
141
+ pass
142
+
143
+ def get(self, topic: str) -> AwknoPage:
144
+ """Get a page by topic name.
145
+
146
+ Args:
147
+ topic: The topic name (case-insensitive).
148
+
149
+ Returns:
150
+ The AwknoPage.
151
+
152
+ Raises:
153
+ NotFoundError: If the topic is not found.
154
+ """
155
+ key = topic.lower()
156
+ if key in self.pages:
157
+ return self.pages[key]
158
+ raise NotFoundError(f"Topic '{topic}' not found")
159
+
160
+ def list_topics(self) -> list[str]:
161
+ """List all available topics."""
162
+ return sorted(self.pages.keys())
163
+
164
+ def list_by_category(self, category: str) -> list[AwknoPage]:
165
+ """List pages by category.
166
+
167
+ Args:
168
+ category: One of "brick", "stack", "law", "topic".
169
+
170
+ Returns:
171
+ List of pages in that category.
172
+ """
173
+ return [
174
+ page for page in self.pages.values() if page.category == category
175
+ ]
176
+
177
+ def search(self, term: str) -> list[tuple[AwknoPage, float]]:
178
+ """Search pages by keyword.
179
+
180
+ Args:
181
+ term: Search term (case-insensitive).
182
+
183
+ Returns:
184
+ List of (page, score) tuples sorted by score (highest first).
185
+ """
186
+ term_lower = term.lower()
187
+ results = []
188
+
189
+ for page in self.pages.values():
190
+ # Score based on where the term appears
191
+ score = 0.0
192
+
193
+ if term_lower in page.topic.lower():
194
+ score += 1.0
195
+
196
+ if page.synopsis and term_lower in page.synopsis.lower():
197
+ score += 0.8
198
+
199
+ if page.description and term_lower in page.description.lower():
200
+ score += 0.5
201
+
202
+ if page.adopt and term_lower in page.adopt.lower():
203
+ score += 0.3
204
+
205
+ if page.body and term_lower in page.body.lower():
206
+ score += 0.2
207
+
208
+ # Use SequenceMatcher for fuzzy matching on topic
209
+ ratio = SequenceMatcher(None, term_lower, page.topic.lower()).ratio()
210
+ score += ratio * 0.5
211
+
212
+ if score > 0:
213
+ results.append((page, score))
214
+
215
+ results.sort(key=lambda x: x[1], reverse=True)
216
+ return results
awkno/generate.py ADDED
@@ -0,0 +1,444 @@
1
+ """Generate the awkno corpus from sources.
2
+
3
+ Run this script to regenerate the offline man-page corpus from ecosystem.yaml
4
+ and the laws in awknowledge/laws/. The output is committed as JSON files
5
+ under awkno/pages/ so the installed package is fully offline.
6
+
7
+ python awkno/generate.py
8
+
9
+ Output files are deterministically ordered and formatted for reproducible
10
+ corpus generation.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import re
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ try:
21
+ import yaml
22
+ except ImportError:
23
+ print("ERROR: PyYAML required for generate.py")
24
+ print("Install with: pip install PyYAML")
25
+ exit(1)
26
+
27
+
28
+ class AwknoGenerator:
29
+ """Generate the corpus from sources."""
30
+
31
+ def __init__(self, repo_root: str | None = None,
32
+ output_dir: str | Path | None = None):
33
+ """Initialize the generator.
34
+
35
+ Args:
36
+ repo_root: Path to repo root (auto-detected if not provided).
37
+ output_dir: Where to write the corpus. Defaults to the committed
38
+ `pages/` beside this file. A caller passes a temp dir to ask
39
+ "is what is committed what I would generate now?" -- a question
40
+ that cannot be answered by a run that overwrites the answer.
41
+ """
42
+ if repo_root is None:
43
+ # Auto-detect: walk up from this file to find the repo root
44
+ current = Path(__file__).parent
45
+ while current != current.parent:
46
+ if (current / "AitherOS" / "config" / "ecosystem.yaml").exists():
47
+ repo_root = str(current / "AitherOS")
48
+ break
49
+ if (current / "config" / "ecosystem.yaml").exists():
50
+ repo_root = str(current)
51
+ break
52
+ current = current.parent
53
+
54
+ if repo_root is None:
55
+ raise ValueError("Could not find repo root. Pass repo_root explicitly.")
56
+
57
+ self.repo_root = Path(repo_root)
58
+ self.ecosystem_file = self.repo_root / "config" / "ecosystem.yaml"
59
+ self.laws_dir = self.repo_root.parent / "awknowledge" / "laws"
60
+ # The Aither World Guide: the journey manifest and its chapter prose.
61
+ # Same brick as the laws, same offline promise -- `awkno guide 2` must
62
+ # answer on a plane with no network, which is why it is COMMITTED here.
63
+ self.guide_dir = self.repo_root.parent / "awknowledge"
64
+ self.output_dir = (Path(output_dir) if output_dir
65
+ else Path(__file__).parent / "pages")
66
+
67
+ if not self.ecosystem_file.exists():
68
+ raise FileNotFoundError(f"ecosystem.yaml not found at {self.ecosystem_file}")
69
+
70
+ def generate(self) -> dict[str, Any]:
71
+ """Generate the full corpus.
72
+
73
+ Returns:
74
+ A dict mapping page keys to their content (for testing).
75
+ """
76
+ self.output_dir.mkdir(parents=True, exist_ok=True)
77
+ pages = {}
78
+
79
+ # Load ecosystem
80
+ # encoding is NOT optional here. Python's default is the platform's
81
+ # preferred encoding, which is cp1252 on Windows — so every em dash in
82
+ # the registry decoded to three junk characters. It never LOOKED wrong,
83
+ # because the writer below escaped non-ASCII on the way out, turning the
84
+ # damage into `—` inside the JSON where no text search
85
+ # for a mojibake sequence would ever match it.
86
+ with open(self.ecosystem_file, encoding="utf-8") as f:
87
+ ecosystem = yaml.safe_load(f)
88
+
89
+ # Generate brick pages
90
+ if "bricks" in ecosystem:
91
+ for brick in ecosystem["bricks"]:
92
+ page = self._make_brick_page(brick)
93
+ pages[page["topic"].lower()] = page
94
+
95
+ # Generate stack pages
96
+ if "stacks" in ecosystem:
97
+ for stack in ecosystem["stacks"]:
98
+ page = self._make_stack_page(stack)
99
+ pages[page["topic"].lower()] = page
100
+
101
+ # Generate law pages
102
+ if self.laws_dir.exists():
103
+ for law_file in sorted(self.laws_dir.glob("*.md")):
104
+ page = self._make_law_page(law_file)
105
+ pages[page["topic"].lower()] = page
106
+
107
+ # Generate guide pages (the journey), plus the guide index itself
108
+ journey = self.guide_dir / "journey.yaml"
109
+ if journey.exists():
110
+ for page in self._make_guide_pages(journey):
111
+ pages[page["topic"].lower()] = page
112
+
113
+ # Write all pages to JSON
114
+ for topic_key, page_data in pages.items():
115
+ output_file = self.output_dir / f"{topic_key}.json"
116
+ # ensure_ascii=False keeps real characters in the file. With the
117
+ # default (True) any encoding damage upstream is escaped into
118
+ # \uXXXX and becomes invisible to inspection — the corpus reads as
119
+ # clean ASCII while rendering as garbage in the terminal.
120
+ with open(output_file, "w", encoding="utf-8") as f:
121
+ json.dump(page_data, f, indent=2, sort_keys=True, ensure_ascii=False)
122
+
123
+ # PRUNE. Writing without deleting makes this a write-ONLY mirror: a brick
124
+ # retired from the registry keeps its page here forever, and `awkno <it>`
125
+ # then answers a stranger, offline, about something that does not exist --
126
+ # which reads exactly like the brick existing. Found live 2026-08-22:
127
+ # `awflow` and `awroll` had been shipping as `status: planned` stubs while
128
+ # appearing in no registry entry and no package. The generator had run
129
+ # correctly every time; nothing it wrote was wrong, and the corpus was
130
+ # still wrong, because the defect was in what it DIDN'T write.
131
+ #
132
+ # Safe because `pages` above is the complete corpus -- bricks, stacks AND
133
+ # laws -- so anything on disk it does not name is by definition stale.
134
+ # Asserted from the other side upstream: the registry's own gate compares
135
+ # this committed corpus against it and fails on drift in EITHER direction,
136
+ # so a regression here cannot pass silently.
137
+ keep = {f"{k}.json" for k in pages}
138
+ removed = []
139
+ for stale in sorted(self.output_dir.glob("*.json")):
140
+ if stale.name not in keep:
141
+ stale.unlink()
142
+ removed.append(stale.stem)
143
+ if removed:
144
+ print(f"[PRUNED] {len(removed)} stale page(s): {', '.join(removed)}")
145
+
146
+ return pages
147
+
148
+ def _scrub_internal_refs(self, text: str) -> str:
149
+ """Remove internal references from text.
150
+
151
+ Removes debt IDs (D-NNNN), rule codes (PQ, EC, SEC, etc), and other
152
+ internal identifiers that should not appear in public documentation.
153
+
154
+ Args:
155
+ text: Text to scrub.
156
+
157
+ Returns:
158
+ Text with internal references removed.
159
+ """
160
+ import re
161
+
162
+ if not text:
163
+ return text
164
+
165
+ # Issue-tracker row ids, bare and bracketed.
166
+ text = re.sub(r"\(\s*D-\d+\s*\)", "", text)
167
+ text = re.sub(r"\[\s*D-\d+\s*\]", "", text)
168
+ text = re.sub(r"\bD-\d+\b", "", text)
169
+
170
+ # Quality-gate rule codes.
171
+ text = re.sub(
172
+ r"\b(?:PQ|EC|SEC|ONB|MRP|SHW|NX|RB|TP|DAW|SAE|CX|MR|AC|ACG|AWM|AWP|QCP|QIC)"
173
+ r"\d{3,4}\b",
174
+ "",
175
+ text,
176
+ )
177
+ text = re.sub(r"gate\s+\d+[a-z]*", "", text)
178
+
179
+ # Names of internal gate scripts. These reach the corpus through the LAW
180
+ # TEXT itself, which cites the checker that caught each defect -- so the
181
+ # two rules above, which only look for ids, passed them straight through.
182
+ # A law keeps all of its force as "a gate that asserted entry points";
183
+ # the filename is internal vocabulary and carries none of the lesson.
184
+ # The replacement must stay a valid FILENAME, because these names appear
185
+ # inside runnable code blocks as well as in prose. Substituting a phrase
186
+ # ("an internal gate") reads fine in a sentence, but inside a shell
187
+ # example it turns a runnable command into an unrunnable one presented
188
+ # as runnable — a worse defect than the disclosure it was fixing. A
189
+ # placeholder FILENAME reads correctly in both places and is obviously a
190
+ # placeholder.
191
+ text = re.sub(r"\bcheck_[a-z0-9_]+\.py\b", "your_checker.py", text)
192
+
193
+ # Collapse whitespace left behind by the removals above, so a scrubbed
194
+ # sentence does not read as though a word is missing.
195
+ text = re.sub(r"[ \t]{2,}", " ", text)
196
+ text = re.sub(r" ([,.;:)])", r"\1", text)
197
+
198
+ return text
199
+
200
+ def _make_brick_page(self, brick: dict) -> dict[str, Any]:
201
+ """Create a page for a brick.
202
+
203
+ Args:
204
+ brick: A brick entry from ecosystem.yaml.
205
+
206
+ Returns:
207
+ A page dict ready for JSON serialization.
208
+ """
209
+ brick_id = brick.get("id", "unknown")
210
+ tagline = brick.get("tagline", "")
211
+ synopsis = tagline.split(" — ")[0] if " — " in tagline else tagline
212
+
213
+ adopt = self._scrub_internal_refs(brick.get("adopt", ""))
214
+ status = brick.get("status", "unknown")
215
+ install = brick.get("install", "")
216
+ problem = self._scrub_internal_refs(brick.get("problem", ""))
217
+ kind = brick.get("kind", "tool")
218
+
219
+ description = f"Kind: {kind}\n"
220
+ if problem:
221
+ description += f"\nProblem\n{problem}\n"
222
+ if install:
223
+ description += f"\nInstall\n{install}"
224
+
225
+ see_also = []
226
+ if brick.get("pairs_with"):
227
+ see_also.extend(brick["pairs_with"])
228
+ if brick.get("includes"):
229
+ see_also.extend(brick["includes"])
230
+
231
+ return {
232
+ "adopt": adopt if adopt else None,
233
+ "category": "brick",
234
+ "description": description.strip(),
235
+ "see_also": see_also if see_also else None,
236
+ "status": status,
237
+ "synopsis": synopsis,
238
+ "topic": brick_id,
239
+ }
240
+
241
+ def _make_stack_page(self, stack: dict) -> dict[str, Any]:
242
+ """Create a page for a stack.
243
+
244
+ Args:
245
+ stack: A stack entry from ecosystem.yaml.
246
+
247
+ Returns:
248
+ A page dict ready for JSON serialization.
249
+ """
250
+ stack_id = stack.get("id", "unknown")
251
+ name = stack.get("name", "")
252
+ what = stack.get("what", "")
253
+ status = stack.get("status", "unknown")
254
+ bricks = stack.get("bricks", [])
255
+
256
+ synopsis = name if name else stack_id
257
+
258
+ # Clean internal references from what field
259
+ description = self._scrub_internal_refs(what)
260
+ description = description + "\n\n"
261
+ if bricks:
262
+ description += f"Includes: {', '.join(bricks)}"
263
+
264
+ return {
265
+ "adopt": None,
266
+ "category": "stack",
267
+ "description": description.strip(),
268
+ "see_also": bricks if bricks else None,
269
+ "status": status,
270
+ "synopsis": synopsis,
271
+ "topic": stack_id,
272
+ }
273
+
274
+ def _make_guide_pages(self, journey_file: Path) -> list[dict[str, Any]]:
275
+ """Pages for the Aither World Guide: one per chapter, plus `guide`.
276
+
277
+ The chapter's DO steps are rendered from the MANIFEST (journey.yaml),
278
+ never from the prose, so what `awkno guide 2` tells a reader to type is
279
+ exactly what the journey gate checked against the kit. The prose's own
280
+ Teach section is carried as the body. Internal refs are scrubbed like
281
+ every other page -- the guide is written public-safe, and this is the
282
+ second line.
283
+ """
284
+ with open(journey_file, encoding="utf-8") as f:
285
+ journey = yaml.safe_load(f) or {}
286
+ chapters = journey.get("chapters") or []
287
+ out: list[dict[str, Any]] = []
288
+ path_dir = journey_file.parent / "path"
289
+ index_lines = []
290
+ for i, ch in enumerate(chapters):
291
+ cid = str(ch.get("id", ""))
292
+ md_file = path_dir / f"{cid}.md"
293
+ teach = ""
294
+ if md_file.exists():
295
+ text = self._scrub_internal_refs(md_file.read_text(encoding="utf-8"))
296
+ m = re.search(r"^## Teach\b.*?$(.*?)(?=^## |\Z)", text, re.M | re.S)
297
+ teach = (m.group(1).strip() if m else "")
298
+ steps = []
299
+ for n, step in enumerate(ch.get("do") or [], 1):
300
+ opt = " (optional)" if step.get("optional") else ""
301
+ label = "type" if step.get("kind") == "prompt" else "$"
302
+ steps.append(f"{n}.{opt} {label} {step.get('cmd', '')}")
303
+ if step.get("expect"):
304
+ steps.append(f" you should see: {step['expect']}")
305
+ if step.get("if_not"):
306
+ steps.append(f" if not: {step['if_not']}")
307
+ dc = ch.get("done_check") or {}
308
+ if dc:
309
+ steps.append("")
310
+ steps.append(f"done when: $ {dc.get('cmd', '')}")
311
+ steps.append(f" you should see: {dc.get('expect', '')}")
312
+ learned = [f"- {x}" for x in (ch.get("learned") or [])]
313
+ body_parts = [teach]
314
+ if steps:
315
+ body_parts += ["", "DO", ""] + steps
316
+ if learned:
317
+ body_parts += ["", "WHAT YOU LEARNED", ""] + learned
318
+ topic = f"guide-{i:02d}"
319
+ index_lines.append(
320
+ f"{topic:<10} {ch.get('title', '')} (~{int(ch.get('minutes', 0))} min)")
321
+ out.append({
322
+ "adopt": None,
323
+ "body": "\n".join(body_parts).strip() or None,
324
+ "category": "guide",
325
+ "description": (f"Chapter {i} of {len(chapters) - 1} -- "
326
+ f"{ch.get('blurb', '')}\n\nAbout {int(ch.get('minutes', 0))} "
327
+ f"minutes. Read online: https://aitherium.github.io/awknowledge/"
328
+ f"path/{cid}.html"),
329
+ "see_also": list(ch.get("bricks") or []) + (
330
+ [f"guide-{i + 1:02d}"] if i + 1 < len(chapters) else []),
331
+ "status": "published",
332
+ "synopsis": str(ch.get("title", "")),
333
+ "topic": topic,
334
+ "slug": cid,
335
+ })
336
+ site = journey.get("site") or {}
337
+ out.append({
338
+ "adopt": "awkno guide 1 # start at chapter 1; `awkno open guide` for the browser",
339
+ "body": "\n".join(index_lines) or None,
340
+ "category": "guide",
341
+ "description": str(site.get("tagline", "The Aither World Guide")) +
342
+ "\n\nOnline: " +
343
+ str(site.get("url", "https://aitherium.github.io/awknowledge/")),
344
+ "see_also": [p["topic"] for p in out],
345
+ "status": "published",
346
+ "synopsis": str(site.get("title", "The Aither World Guide")),
347
+ "topic": "guide",
348
+ "slug": None,
349
+ })
350
+ return out
351
+
352
+ def _make_law_page(self, law_file: Path) -> dict[str, Any]:
353
+ """Create a page for a law.
354
+
355
+ Args:
356
+ law_file: Path to a law markdown file.
357
+
358
+ Returns:
359
+ A page dict ready for JSON serialization.
360
+ """
361
+ # Extract law number and slug from filename: 01-rule-name.md
362
+ stem = law_file.stem
363
+ parts = stem.split("-", 1)
364
+ law_num = int(parts[0])
365
+ law_slug = parts[1] if len(parts) > 1 else ""
366
+
367
+ # Read the file, and SCRUB IT — the law bodies are the longest prose in
368
+ # this corpus and the only source that cites internal gate scripts by
369
+ # name. Brick and stack pages were scrubbed from the start; this call
370
+ # site was missed, so a correct sanitiser shipped internal identifiers
371
+ # from three law pages while every other page was clean.
372
+ content = self._scrub_internal_refs(law_file.read_text(encoding="utf-8"))
373
+
374
+ # Extract the title (first line, usually # heading)
375
+ lines = content.strip().split("\n")
376
+ title = ""
377
+ description = ""
378
+
379
+ for i, line in enumerate(lines):
380
+ if line.startswith("# "):
381
+ title = line[2:].strip()
382
+ description = "\n".join(lines[i + 1 :]).strip()
383
+ break
384
+
385
+ # Use the title as the synopsis
386
+ synopsis = title if title else law_slug.replace("-", " ").title()
387
+
388
+ # Create topic key as "law-01" for law 1
389
+ topic_key = f"law-{law_num:02d}"
390
+
391
+ return {
392
+ "adopt": None,
393
+ "body": description if description else None,
394
+ "category": "law",
395
+ "description": f"Law #{law_num}",
396
+ "see_also": None,
397
+ "status": "published",
398
+ "synopsis": synopsis,
399
+ "topic": topic_key,
400
+ # The filename slug. `awkno law <N|SLUG>` advertised this lookup
401
+ # while the value was computed here and thrown away, so no slug
402
+ # ever resolved -- a documented invocation that could not work.
403
+ "slug": law_slug or None,
404
+ }
405
+
406
+
407
+ def main() -> None:
408
+ """CLI entry point for corpus generation."""
409
+ import sys
410
+
411
+ # --out DIR writes elsewhere, leaving the committed corpus untouched. Parsed by
412
+ # hand rather than with argparse to keep the positional repo_root working exactly
413
+ # as it did -- this is a published module, and a CLI that changes shape under a
414
+ # caller is a breakage nobody attributes to a "docs" change.
415
+ argv = sys.argv[1:]
416
+ out_dir = None
417
+ if "--out" in argv:
418
+ i = argv.index("--out")
419
+ if i + 1 >= len(argv):
420
+ print("ERROR: --out needs a directory")
421
+ exit(2)
422
+ out_dir = argv[i + 1]
423
+ argv = argv[:i] + argv[i + 2:]
424
+
425
+ try:
426
+ # Allow repo root to be passed as first argument
427
+ repo_root = argv[0] if argv else None
428
+ generator = AwknoGenerator(repo_root, output_dir=out_dir)
429
+ pages = generator.generate()
430
+ print(f"[OK] Generated corpus with {len(pages)} pages")
431
+ print(f" Output: {generator.output_dir}")
432
+
433
+ except FileNotFoundError as e:
434
+ print(f"ERROR: {e}")
435
+ exit(1)
436
+ except ValueError as e:
437
+ print(f"ERROR: {e}")
438
+ print("\nUsage: python awkno/generate.py [repo_root]")
439
+ print("\nIf repo_root is not provided, auto-detects from ecosystem.yaml location")
440
+ exit(1)
441
+
442
+
443
+ if __name__ == "__main__":
444
+ main()
@@ -0,0 +1,13 @@
1
+ {
2
+ "adopt": null,
3
+ "category": "stack",
4
+ "description": "What an agent needs to perceive things outside its own process — search and pages.\n\nIncludes: awfind, awbrowse, awdk",
5
+ "see_also": [
6
+ "awfind",
7
+ "awbrowse",
8
+ "awdk"
9
+ ],
10
+ "status": "planned",
11
+ "synopsis": "Senses",
12
+ "topic": "agent-senses"
13
+ }
@@ -0,0 +1,14 @@
1
+ {
2
+ "adopt": null,
3
+ "category": "stack",
4
+ "description": "awnix already carries the capabilities; adding an agent is three lines in a Dockerfile, and your skills and credentials layer on top of that. The point of the order is that the guarantees live in the base, so swapping the agent keeps them.\n\nIncludes: awnix, awdk, awskills, awkno",
5
+ "see_also": [
6
+ "awnix",
7
+ "awdk",
8
+ "awskills",
9
+ "awkno"
10
+ ],
11
+ "status": "partial",
12
+ "synopsis": "The bare agent VM",
13
+ "topic": "agent-vm"
14
+ }