awkno 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- awkno/__init__.py +25 -0
- awkno/_doctor.py +114 -0
- awkno/cli.py +497 -0
- awkno/corpus.py +216 -0
- awkno/generate.py +444 -0
- awkno/pages/agent-senses.json +13 -0
- awkno/pages/agent-vm.json +14 -0
- awkno/pages/aitherconnect.json +11 -0
- awkno/pages/aitherkvcache.json +11 -0
- awkno/pages/aitherzero.json +11 -0
- awkno/pages/awarena.json +14 -0
- awkno/pages/awask.json +15 -0
- awkno/pages/awbac.json +11 -0
- awkno/pages/awbrowse.json +13 -0
- awkno/pages/awdit.json +12 -0
- awkno/pages/awdk.json +16 -0
- awkno/pages/awevolve.json +15 -0
- awkno/pages/awfind.json +13 -0
- awkno/pages/awgit.json +12 -0
- awkno/pages/awgraph.json +12 -0
- awkno/pages/awiam.json +12 -0
- awkno/pages/awkit.json +13 -0
- awkno/pages/awkno.json +13 -0
- awkno/pages/awknowledge.json +12 -0
- awkno/pages/awm.json +13 -0
- awkno/pages/awmail.json +18 -0
- awkno/pages/awnboard.json +20 -0
- awkno/pages/awnest.json +19 -0
- awkno/pages/awnet.json +13 -0
- awkno/pages/awnix.json +20 -0
- awkno/pages/awnode.json +12 -0
- awkno/pages/awpack.json +13 -0
- awkno/pages/awpredict.json +11 -0
- awkno/pages/awprism.json +14 -0
- awkno/pages/awreason.json +14 -0
- awkno/pages/awrecover.json +13 -0
- awkno/pages/awrecurse.json +13 -0
- awkno/pages/awrelay.json +12 -0
- awkno/pages/awrepl.json +13 -0
- awkno/pages/awresearch.json +14 -0
- awkno/pages/awrun.json +13 -0
- awkno/pages/awseal.json +12 -0
- awkno/pages/awsh.json +13 -0
- awkno/pages/awshare.json +12 -0
- awkno/pages/awskills.json +12 -0
- awkno/pages/awsync.json +16 -0
- awkno/pages/awtunnel.json +12 -0
- awkno/pages/cited-research.json +14 -0
- awkno/pages/gobbonet-agentic.json +14 -0
- awkno/pages/guide-00.json +16 -0
- awkno/pages/guide-01.json +14 -0
- awkno/pages/guide-02.json +14 -0
- awkno/pages/guide-03.json +14 -0
- awkno/pages/guide-04.json +15 -0
- awkno/pages/guide-05.json +16 -0
- awkno/pages/guide-06.json +15 -0
- awkno/pages/guide-07.json +15 -0
- awkno/pages/guide-08.json +17 -0
- awkno/pages/guide-09.json +13 -0
- awkno/pages/guide.json +22 -0
- awkno/pages/law-01.json +11 -0
- awkno/pages/law-02.json +11 -0
- awkno/pages/law-03.json +11 -0
- awkno/pages/law-04.json +11 -0
- awkno/pages/law-05.json +11 -0
- awkno/pages/law-06.json +11 -0
- awkno/pages/law-07.json +11 -0
- awkno/pages/law-08.json +11 -0
- awkno/pages/law-09.json +11 -0
- awkno/pages/law-10.json +11 -0
- awkno/pages/law-11.json +11 -0
- awkno/pages/law-12.json +11 -0
- awkno/pages/law-13.json +11 -0
- awkno/pages/law-14.json +11 -0
- awkno/pages/law-15.json +11 -0
- awkno/pages/law-16.json +11 -0
- awkno/pages/law-17.json +11 -0
- awkno/pages/law-18.json +11 -0
- awkno/pages/law-19.json +11 -0
- awkno/pages/one-surface.json +15 -0
- awkno/pages/provenance.json +14 -0
- awkno/pages/shared-worktree.json +14 -0
- awkno/pages/the-front-door.json +15 -0
- awkno/pages/the-reasoning-loop.json +14 -0
- awkno/pages/who-what-did.json +13 -0
- awkno-0.2.0.dist-info/METADATA +321 -0
- awkno-0.2.0.dist-info/RECORD +91 -0
- awkno-0.2.0.dist-info/WHEEL +5 -0
- awkno-0.2.0.dist-info/entry_points.txt +2 -0
- awkno-0.2.0.dist-info/licenses/LICENSE +118 -0
- awkno-0.2.0.dist-info/top_level.txt +1 -0
awkno/corpus.py
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
"""Load and query the offline corpus.
|
|
2
|
+
|
|
3
|
+
Pages are stored as JSON files under awkno/pages/. This module handles the
|
|
4
|
+
loading, searching, and formatting of those pages.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from difflib import SequenceMatcher
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Optional
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class NotFoundError(Exception):
|
|
17
|
+
"""Raised when a topic is not found."""
|
|
18
|
+
|
|
19
|
+
pass
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class AwknoPage:
|
|
24
|
+
"""A single man page."""
|
|
25
|
+
|
|
26
|
+
topic: str
|
|
27
|
+
category: str
|
|
28
|
+
synopsis: str
|
|
29
|
+
description: str
|
|
30
|
+
adopt: Optional[str] = None
|
|
31
|
+
status: Optional[str] = None
|
|
32
|
+
see_also: Optional[list[str]] = None
|
|
33
|
+
body: Optional[str] = None
|
|
34
|
+
# The law filename's own slug (`design-for-the-silence`). Carried rather
|
|
35
|
+
# than re-derived from the synopsis: the generator already computes it, and
|
|
36
|
+
# a slug inferred from a title silently stops matching the moment a law is
|
|
37
|
+
# retitled.
|
|
38
|
+
slug: Optional[str] = None
|
|
39
|
+
|
|
40
|
+
def render(self, plain: bool = False) -> str:
|
|
41
|
+
"""Render the page as text.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
plain: If True, no ANSI codes. If False, use formatting (when TTY).
|
|
45
|
+
|
|
46
|
+
Returns:
|
|
47
|
+
The formatted page text.
|
|
48
|
+
"""
|
|
49
|
+
lines = []
|
|
50
|
+
lines.append("NAME")
|
|
51
|
+
lines.append(f" {self.topic} — {self.synopsis}")
|
|
52
|
+
lines.append("")
|
|
53
|
+
|
|
54
|
+
if self.adopt:
|
|
55
|
+
lines.append("ADOPT")
|
|
56
|
+
for line in self.adopt.split("\n"):
|
|
57
|
+
lines.append(f" {line}")
|
|
58
|
+
lines.append("")
|
|
59
|
+
|
|
60
|
+
if self.description:
|
|
61
|
+
lines.append("DESCRIPTION")
|
|
62
|
+
for line in self.description.split("\n"):
|
|
63
|
+
lines.append(f" {line}")
|
|
64
|
+
lines.append("")
|
|
65
|
+
|
|
66
|
+
if self.body:
|
|
67
|
+
lines.append("DETAILS")
|
|
68
|
+
for line in self.body.split("\n"):
|
|
69
|
+
lines.append(f" {line}")
|
|
70
|
+
lines.append("")
|
|
71
|
+
|
|
72
|
+
if self.status:
|
|
73
|
+
lines.append("STATUS")
|
|
74
|
+
lines.append(f" {self.status}")
|
|
75
|
+
lines.append("")
|
|
76
|
+
|
|
77
|
+
if self.see_also:
|
|
78
|
+
lines.append("SEE ALSO")
|
|
79
|
+
for item in self.see_also:
|
|
80
|
+
lines.append(f" {item}")
|
|
81
|
+
|
|
82
|
+
text = "\n".join(lines)
|
|
83
|
+
|
|
84
|
+
if not plain:
|
|
85
|
+
# Check if stdout is a TTY for formatting
|
|
86
|
+
import sys
|
|
87
|
+
|
|
88
|
+
if hasattr(sys.stdout, "isatty") and sys.stdout.isatty():
|
|
89
|
+
# Apply ANSI formatting
|
|
90
|
+
text = text.replace("NAME\n", "\033[1mNAME\033[0m\n")
|
|
91
|
+
text = text.replace("ADOPT\n", "\033[1mADOPT\033[0m\n")
|
|
92
|
+
text = text.replace("DESCRIPTION\n", "\033[1mDESCRIPTION\033[0m\n")
|
|
93
|
+
text = text.replace("DETAILS\n", "\033[1mDETAILS\033[0m\n")
|
|
94
|
+
text = text.replace("STATUS\n", "\033[1mSTATUS\033[0m\n")
|
|
95
|
+
text = text.replace("SEE ALSO\n", "\033[1mSEE ALSO\033[0m\n")
|
|
96
|
+
|
|
97
|
+
return text
|
|
98
|
+
|
|
99
|
+
def to_dict(self) -> dict:
|
|
100
|
+
"""Convert to dict for JSON serialization."""
|
|
101
|
+
return {
|
|
102
|
+
"topic": self.topic,
|
|
103
|
+
"category": self.category,
|
|
104
|
+
"synopsis": self.synopsis,
|
|
105
|
+
"description": self.description,
|
|
106
|
+
"adopt": self.adopt,
|
|
107
|
+
"status": self.status,
|
|
108
|
+
"see_also": self.see_also,
|
|
109
|
+
"body": self.body,
|
|
110
|
+
"slug": self.slug,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
@classmethod
|
|
114
|
+
def from_dict(cls, data: dict) -> AwknoPage:
|
|
115
|
+
"""Create from dict (JSON deserialization)."""
|
|
116
|
+
return cls(**data)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class AwknoRegistry:
|
|
120
|
+
"""Load and query all pages."""
|
|
121
|
+
|
|
122
|
+
def __init__(self):
|
|
123
|
+
"""Load all pages from the corpus directory."""
|
|
124
|
+
self.pages: dict[str, AwknoPage] = {}
|
|
125
|
+
self._load_pages()
|
|
126
|
+
|
|
127
|
+
def _load_pages(self) -> None:
|
|
128
|
+
"""Load all JSON files from awkno/pages/."""
|
|
129
|
+
pages_dir = Path(__file__).parent / "pages"
|
|
130
|
+
|
|
131
|
+
if not pages_dir.exists():
|
|
132
|
+
return
|
|
133
|
+
|
|
134
|
+
for json_file in pages_dir.glob("*.json"):
|
|
135
|
+
try:
|
|
136
|
+
with open(json_file) as f:
|
|
137
|
+
data = json.load(f)
|
|
138
|
+
page = AwknoPage.from_dict(data)
|
|
139
|
+
self.pages[page.topic.lower()] = page
|
|
140
|
+
except (json.JSONDecodeError, ValueError):
|
|
141
|
+
pass
|
|
142
|
+
|
|
143
|
+
def get(self, topic: str) -> AwknoPage:
|
|
144
|
+
"""Get a page by topic name.
|
|
145
|
+
|
|
146
|
+
Args:
|
|
147
|
+
topic: The topic name (case-insensitive).
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
The AwknoPage.
|
|
151
|
+
|
|
152
|
+
Raises:
|
|
153
|
+
NotFoundError: If the topic is not found.
|
|
154
|
+
"""
|
|
155
|
+
key = topic.lower()
|
|
156
|
+
if key in self.pages:
|
|
157
|
+
return self.pages[key]
|
|
158
|
+
raise NotFoundError(f"Topic '{topic}' not found")
|
|
159
|
+
|
|
160
|
+
def list_topics(self) -> list[str]:
|
|
161
|
+
"""List all available topics."""
|
|
162
|
+
return sorted(self.pages.keys())
|
|
163
|
+
|
|
164
|
+
def list_by_category(self, category: str) -> list[AwknoPage]:
|
|
165
|
+
"""List pages by category.
|
|
166
|
+
|
|
167
|
+
Args:
|
|
168
|
+
category: One of "brick", "stack", "law", "topic".
|
|
169
|
+
|
|
170
|
+
Returns:
|
|
171
|
+
List of pages in that category.
|
|
172
|
+
"""
|
|
173
|
+
return [
|
|
174
|
+
page for page in self.pages.values() if page.category == category
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
def search(self, term: str) -> list[tuple[AwknoPage, float]]:
|
|
178
|
+
"""Search pages by keyword.
|
|
179
|
+
|
|
180
|
+
Args:
|
|
181
|
+
term: Search term (case-insensitive).
|
|
182
|
+
|
|
183
|
+
Returns:
|
|
184
|
+
List of (page, score) tuples sorted by score (highest first).
|
|
185
|
+
"""
|
|
186
|
+
term_lower = term.lower()
|
|
187
|
+
results = []
|
|
188
|
+
|
|
189
|
+
for page in self.pages.values():
|
|
190
|
+
# Score based on where the term appears
|
|
191
|
+
score = 0.0
|
|
192
|
+
|
|
193
|
+
if term_lower in page.topic.lower():
|
|
194
|
+
score += 1.0
|
|
195
|
+
|
|
196
|
+
if page.synopsis and term_lower in page.synopsis.lower():
|
|
197
|
+
score += 0.8
|
|
198
|
+
|
|
199
|
+
if page.description and term_lower in page.description.lower():
|
|
200
|
+
score += 0.5
|
|
201
|
+
|
|
202
|
+
if page.adopt and term_lower in page.adopt.lower():
|
|
203
|
+
score += 0.3
|
|
204
|
+
|
|
205
|
+
if page.body and term_lower in page.body.lower():
|
|
206
|
+
score += 0.2
|
|
207
|
+
|
|
208
|
+
# Use SequenceMatcher for fuzzy matching on topic
|
|
209
|
+
ratio = SequenceMatcher(None, term_lower, page.topic.lower()).ratio()
|
|
210
|
+
score += ratio * 0.5
|
|
211
|
+
|
|
212
|
+
if score > 0:
|
|
213
|
+
results.append((page, score))
|
|
214
|
+
|
|
215
|
+
results.sort(key=lambda x: x[1], reverse=True)
|
|
216
|
+
return results
|
awkno/generate.py
ADDED
|
@@ -0,0 +1,444 @@
|
|
|
1
|
+
"""Generate the awkno corpus from sources.
|
|
2
|
+
|
|
3
|
+
Run this script to regenerate the offline man-page corpus from ecosystem.yaml
|
|
4
|
+
and the laws in awknowledge/laws/. The output is committed as JSON files
|
|
5
|
+
under awkno/pages/ so the installed package is fully offline.
|
|
6
|
+
|
|
7
|
+
python awkno/generate.py
|
|
8
|
+
|
|
9
|
+
Output files are deterministically ordered and formatted for reproducible
|
|
10
|
+
corpus generation.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import re
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
try:
|
|
21
|
+
import yaml
|
|
22
|
+
except ImportError:
|
|
23
|
+
print("ERROR: PyYAML required for generate.py")
|
|
24
|
+
print("Install with: pip install PyYAML")
|
|
25
|
+
exit(1)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class AwknoGenerator:
|
|
29
|
+
"""Generate the corpus from sources."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, repo_root: str | None = None,
|
|
32
|
+
output_dir: str | Path | None = None):
|
|
33
|
+
"""Initialize the generator.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
repo_root: Path to repo root (auto-detected if not provided).
|
|
37
|
+
output_dir: Where to write the corpus. Defaults to the committed
|
|
38
|
+
`pages/` beside this file. A caller passes a temp dir to ask
|
|
39
|
+
"is what is committed what I would generate now?" -- a question
|
|
40
|
+
that cannot be answered by a run that overwrites the answer.
|
|
41
|
+
"""
|
|
42
|
+
if repo_root is None:
|
|
43
|
+
# Auto-detect: walk up from this file to find the repo root
|
|
44
|
+
current = Path(__file__).parent
|
|
45
|
+
while current != current.parent:
|
|
46
|
+
if (current / "AitherOS" / "config" / "ecosystem.yaml").exists():
|
|
47
|
+
repo_root = str(current / "AitherOS")
|
|
48
|
+
break
|
|
49
|
+
if (current / "config" / "ecosystem.yaml").exists():
|
|
50
|
+
repo_root = str(current)
|
|
51
|
+
break
|
|
52
|
+
current = current.parent
|
|
53
|
+
|
|
54
|
+
if repo_root is None:
|
|
55
|
+
raise ValueError("Could not find repo root. Pass repo_root explicitly.")
|
|
56
|
+
|
|
57
|
+
self.repo_root = Path(repo_root)
|
|
58
|
+
self.ecosystem_file = self.repo_root / "config" / "ecosystem.yaml"
|
|
59
|
+
self.laws_dir = self.repo_root.parent / "awknowledge" / "laws"
|
|
60
|
+
# The Aither World Guide: the journey manifest and its chapter prose.
|
|
61
|
+
# Same brick as the laws, same offline promise -- `awkno guide 2` must
|
|
62
|
+
# answer on a plane with no network, which is why it is COMMITTED here.
|
|
63
|
+
self.guide_dir = self.repo_root.parent / "awknowledge"
|
|
64
|
+
self.output_dir = (Path(output_dir) if output_dir
|
|
65
|
+
else Path(__file__).parent / "pages")
|
|
66
|
+
|
|
67
|
+
if not self.ecosystem_file.exists():
|
|
68
|
+
raise FileNotFoundError(f"ecosystem.yaml not found at {self.ecosystem_file}")
|
|
69
|
+
|
|
70
|
+
def generate(self) -> dict[str, Any]:
|
|
71
|
+
"""Generate the full corpus.
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
A dict mapping page keys to their content (for testing).
|
|
75
|
+
"""
|
|
76
|
+
self.output_dir.mkdir(parents=True, exist_ok=True)
|
|
77
|
+
pages = {}
|
|
78
|
+
|
|
79
|
+
# Load ecosystem
|
|
80
|
+
# encoding is NOT optional here. Python's default is the platform's
|
|
81
|
+
# preferred encoding, which is cp1252 on Windows — so every em dash in
|
|
82
|
+
# the registry decoded to three junk characters. It never LOOKED wrong,
|
|
83
|
+
# because the writer below escaped non-ASCII on the way out, turning the
|
|
84
|
+
# damage into `—` inside the JSON where no text search
|
|
85
|
+
# for a mojibake sequence would ever match it.
|
|
86
|
+
with open(self.ecosystem_file, encoding="utf-8") as f:
|
|
87
|
+
ecosystem = yaml.safe_load(f)
|
|
88
|
+
|
|
89
|
+
# Generate brick pages
|
|
90
|
+
if "bricks" in ecosystem:
|
|
91
|
+
for brick in ecosystem["bricks"]:
|
|
92
|
+
page = self._make_brick_page(brick)
|
|
93
|
+
pages[page["topic"].lower()] = page
|
|
94
|
+
|
|
95
|
+
# Generate stack pages
|
|
96
|
+
if "stacks" in ecosystem:
|
|
97
|
+
for stack in ecosystem["stacks"]:
|
|
98
|
+
page = self._make_stack_page(stack)
|
|
99
|
+
pages[page["topic"].lower()] = page
|
|
100
|
+
|
|
101
|
+
# Generate law pages
|
|
102
|
+
if self.laws_dir.exists():
|
|
103
|
+
for law_file in sorted(self.laws_dir.glob("*.md")):
|
|
104
|
+
page = self._make_law_page(law_file)
|
|
105
|
+
pages[page["topic"].lower()] = page
|
|
106
|
+
|
|
107
|
+
# Generate guide pages (the journey), plus the guide index itself
|
|
108
|
+
journey = self.guide_dir / "journey.yaml"
|
|
109
|
+
if journey.exists():
|
|
110
|
+
for page in self._make_guide_pages(journey):
|
|
111
|
+
pages[page["topic"].lower()] = page
|
|
112
|
+
|
|
113
|
+
# Write all pages to JSON
|
|
114
|
+
for topic_key, page_data in pages.items():
|
|
115
|
+
output_file = self.output_dir / f"{topic_key}.json"
|
|
116
|
+
# ensure_ascii=False keeps real characters in the file. With the
|
|
117
|
+
# default (True) any encoding damage upstream is escaped into
|
|
118
|
+
# \uXXXX and becomes invisible to inspection — the corpus reads as
|
|
119
|
+
# clean ASCII while rendering as garbage in the terminal.
|
|
120
|
+
with open(output_file, "w", encoding="utf-8") as f:
|
|
121
|
+
json.dump(page_data, f, indent=2, sort_keys=True, ensure_ascii=False)
|
|
122
|
+
|
|
123
|
+
# PRUNE. Writing without deleting makes this a write-ONLY mirror: a brick
|
|
124
|
+
# retired from the registry keeps its page here forever, and `awkno <it>`
|
|
125
|
+
# then answers a stranger, offline, about something that does not exist --
|
|
126
|
+
# which reads exactly like the brick existing. Found live 2026-08-22:
|
|
127
|
+
# `awflow` and `awroll` had been shipping as `status: planned` stubs while
|
|
128
|
+
# appearing in no registry entry and no package. The generator had run
|
|
129
|
+
# correctly every time; nothing it wrote was wrong, and the corpus was
|
|
130
|
+
# still wrong, because the defect was in what it DIDN'T write.
|
|
131
|
+
#
|
|
132
|
+
# Safe because `pages` above is the complete corpus -- bricks, stacks AND
|
|
133
|
+
# laws -- so anything on disk it does not name is by definition stale.
|
|
134
|
+
# Asserted from the other side upstream: the registry's own gate compares
|
|
135
|
+
# this committed corpus against it and fails on drift in EITHER direction,
|
|
136
|
+
# so a regression here cannot pass silently.
|
|
137
|
+
keep = {f"{k}.json" for k in pages}
|
|
138
|
+
removed = []
|
|
139
|
+
for stale in sorted(self.output_dir.glob("*.json")):
|
|
140
|
+
if stale.name not in keep:
|
|
141
|
+
stale.unlink()
|
|
142
|
+
removed.append(stale.stem)
|
|
143
|
+
if removed:
|
|
144
|
+
print(f"[PRUNED] {len(removed)} stale page(s): {', '.join(removed)}")
|
|
145
|
+
|
|
146
|
+
return pages
|
|
147
|
+
|
|
148
|
+
def _scrub_internal_refs(self, text: str) -> str:
|
|
149
|
+
"""Remove internal references from text.
|
|
150
|
+
|
|
151
|
+
Removes debt IDs (D-NNNN), rule codes (PQ, EC, SEC, etc), and other
|
|
152
|
+
internal identifiers that should not appear in public documentation.
|
|
153
|
+
|
|
154
|
+
Args:
|
|
155
|
+
text: Text to scrub.
|
|
156
|
+
|
|
157
|
+
Returns:
|
|
158
|
+
Text with internal references removed.
|
|
159
|
+
"""
|
|
160
|
+
import re
|
|
161
|
+
|
|
162
|
+
if not text:
|
|
163
|
+
return text
|
|
164
|
+
|
|
165
|
+
# Issue-tracker row ids, bare and bracketed.
|
|
166
|
+
text = re.sub(r"\(\s*D-\d+\s*\)", "", text)
|
|
167
|
+
text = re.sub(r"\[\s*D-\d+\s*\]", "", text)
|
|
168
|
+
text = re.sub(r"\bD-\d+\b", "", text)
|
|
169
|
+
|
|
170
|
+
# Quality-gate rule codes.
|
|
171
|
+
text = re.sub(
|
|
172
|
+
r"\b(?:PQ|EC|SEC|ONB|MRP|SHW|NX|RB|TP|DAW|SAE|CX|MR|AC|ACG|AWM|AWP|QCP|QIC)"
|
|
173
|
+
r"\d{3,4}\b",
|
|
174
|
+
"",
|
|
175
|
+
text,
|
|
176
|
+
)
|
|
177
|
+
text = re.sub(r"gate\s+\d+[a-z]*", "", text)
|
|
178
|
+
|
|
179
|
+
# Names of internal gate scripts. These reach the corpus through the LAW
|
|
180
|
+
# TEXT itself, which cites the checker that caught each defect -- so the
|
|
181
|
+
# two rules above, which only look for ids, passed them straight through.
|
|
182
|
+
# A law keeps all of its force as "a gate that asserted entry points";
|
|
183
|
+
# the filename is internal vocabulary and carries none of the lesson.
|
|
184
|
+
# The replacement must stay a valid FILENAME, because these names appear
|
|
185
|
+
# inside runnable code blocks as well as in prose. Substituting a phrase
|
|
186
|
+
# ("an internal gate") reads fine in a sentence, but inside a shell
|
|
187
|
+
# example it turns a runnable command into an unrunnable one presented
|
|
188
|
+
# as runnable — a worse defect than the disclosure it was fixing. A
|
|
189
|
+
# placeholder FILENAME reads correctly in both places and is obviously a
|
|
190
|
+
# placeholder.
|
|
191
|
+
text = re.sub(r"\bcheck_[a-z0-9_]+\.py\b", "your_checker.py", text)
|
|
192
|
+
|
|
193
|
+
# Collapse whitespace left behind by the removals above, so a scrubbed
|
|
194
|
+
# sentence does not read as though a word is missing.
|
|
195
|
+
text = re.sub(r"[ \t]{2,}", " ", text)
|
|
196
|
+
text = re.sub(r" ([,.;:)])", r"\1", text)
|
|
197
|
+
|
|
198
|
+
return text
|
|
199
|
+
|
|
200
|
+
def _make_brick_page(self, brick: dict) -> dict[str, Any]:
|
|
201
|
+
"""Create a page for a brick.
|
|
202
|
+
|
|
203
|
+
Args:
|
|
204
|
+
brick: A brick entry from ecosystem.yaml.
|
|
205
|
+
|
|
206
|
+
Returns:
|
|
207
|
+
A page dict ready for JSON serialization.
|
|
208
|
+
"""
|
|
209
|
+
brick_id = brick.get("id", "unknown")
|
|
210
|
+
tagline = brick.get("tagline", "")
|
|
211
|
+
synopsis = tagline.split(" — ")[0] if " — " in tagline else tagline
|
|
212
|
+
|
|
213
|
+
adopt = self._scrub_internal_refs(brick.get("adopt", ""))
|
|
214
|
+
status = brick.get("status", "unknown")
|
|
215
|
+
install = brick.get("install", "")
|
|
216
|
+
problem = self._scrub_internal_refs(brick.get("problem", ""))
|
|
217
|
+
kind = brick.get("kind", "tool")
|
|
218
|
+
|
|
219
|
+
description = f"Kind: {kind}\n"
|
|
220
|
+
if problem:
|
|
221
|
+
description += f"\nProblem\n{problem}\n"
|
|
222
|
+
if install:
|
|
223
|
+
description += f"\nInstall\n{install}"
|
|
224
|
+
|
|
225
|
+
see_also = []
|
|
226
|
+
if brick.get("pairs_with"):
|
|
227
|
+
see_also.extend(brick["pairs_with"])
|
|
228
|
+
if brick.get("includes"):
|
|
229
|
+
see_also.extend(brick["includes"])
|
|
230
|
+
|
|
231
|
+
return {
|
|
232
|
+
"adopt": adopt if adopt else None,
|
|
233
|
+
"category": "brick",
|
|
234
|
+
"description": description.strip(),
|
|
235
|
+
"see_also": see_also if see_also else None,
|
|
236
|
+
"status": status,
|
|
237
|
+
"synopsis": synopsis,
|
|
238
|
+
"topic": brick_id,
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
def _make_stack_page(self, stack: dict) -> dict[str, Any]:
|
|
242
|
+
"""Create a page for a stack.
|
|
243
|
+
|
|
244
|
+
Args:
|
|
245
|
+
stack: A stack entry from ecosystem.yaml.
|
|
246
|
+
|
|
247
|
+
Returns:
|
|
248
|
+
A page dict ready for JSON serialization.
|
|
249
|
+
"""
|
|
250
|
+
stack_id = stack.get("id", "unknown")
|
|
251
|
+
name = stack.get("name", "")
|
|
252
|
+
what = stack.get("what", "")
|
|
253
|
+
status = stack.get("status", "unknown")
|
|
254
|
+
bricks = stack.get("bricks", [])
|
|
255
|
+
|
|
256
|
+
synopsis = name if name else stack_id
|
|
257
|
+
|
|
258
|
+
# Clean internal references from what field
|
|
259
|
+
description = self._scrub_internal_refs(what)
|
|
260
|
+
description = description + "\n\n"
|
|
261
|
+
if bricks:
|
|
262
|
+
description += f"Includes: {', '.join(bricks)}"
|
|
263
|
+
|
|
264
|
+
return {
|
|
265
|
+
"adopt": None,
|
|
266
|
+
"category": "stack",
|
|
267
|
+
"description": description.strip(),
|
|
268
|
+
"see_also": bricks if bricks else None,
|
|
269
|
+
"status": status,
|
|
270
|
+
"synopsis": synopsis,
|
|
271
|
+
"topic": stack_id,
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
def _make_guide_pages(self, journey_file: Path) -> list[dict[str, Any]]:
|
|
275
|
+
"""Pages for the Aither World Guide: one per chapter, plus `guide`.
|
|
276
|
+
|
|
277
|
+
The chapter's DO steps are rendered from the MANIFEST (journey.yaml),
|
|
278
|
+
never from the prose, so what `awkno guide 2` tells a reader to type is
|
|
279
|
+
exactly what the journey gate checked against the kit. The prose's own
|
|
280
|
+
Teach section is carried as the body. Internal refs are scrubbed like
|
|
281
|
+
every other page -- the guide is written public-safe, and this is the
|
|
282
|
+
second line.
|
|
283
|
+
"""
|
|
284
|
+
with open(journey_file, encoding="utf-8") as f:
|
|
285
|
+
journey = yaml.safe_load(f) or {}
|
|
286
|
+
chapters = journey.get("chapters") or []
|
|
287
|
+
out: list[dict[str, Any]] = []
|
|
288
|
+
path_dir = journey_file.parent / "path"
|
|
289
|
+
index_lines = []
|
|
290
|
+
for i, ch in enumerate(chapters):
|
|
291
|
+
cid = str(ch.get("id", ""))
|
|
292
|
+
md_file = path_dir / f"{cid}.md"
|
|
293
|
+
teach = ""
|
|
294
|
+
if md_file.exists():
|
|
295
|
+
text = self._scrub_internal_refs(md_file.read_text(encoding="utf-8"))
|
|
296
|
+
m = re.search(r"^## Teach\b.*?$(.*?)(?=^## |\Z)", text, re.M | re.S)
|
|
297
|
+
teach = (m.group(1).strip() if m else "")
|
|
298
|
+
steps = []
|
|
299
|
+
for n, step in enumerate(ch.get("do") or [], 1):
|
|
300
|
+
opt = " (optional)" if step.get("optional") else ""
|
|
301
|
+
label = "type" if step.get("kind") == "prompt" else "$"
|
|
302
|
+
steps.append(f"{n}.{opt} {label} {step.get('cmd', '')}")
|
|
303
|
+
if step.get("expect"):
|
|
304
|
+
steps.append(f" you should see: {step['expect']}")
|
|
305
|
+
if step.get("if_not"):
|
|
306
|
+
steps.append(f" if not: {step['if_not']}")
|
|
307
|
+
dc = ch.get("done_check") or {}
|
|
308
|
+
if dc:
|
|
309
|
+
steps.append("")
|
|
310
|
+
steps.append(f"done when: $ {dc.get('cmd', '')}")
|
|
311
|
+
steps.append(f" you should see: {dc.get('expect', '')}")
|
|
312
|
+
learned = [f"- {x}" for x in (ch.get("learned") or [])]
|
|
313
|
+
body_parts = [teach]
|
|
314
|
+
if steps:
|
|
315
|
+
body_parts += ["", "DO", ""] + steps
|
|
316
|
+
if learned:
|
|
317
|
+
body_parts += ["", "WHAT YOU LEARNED", ""] + learned
|
|
318
|
+
topic = f"guide-{i:02d}"
|
|
319
|
+
index_lines.append(
|
|
320
|
+
f"{topic:<10} {ch.get('title', '')} (~{int(ch.get('minutes', 0))} min)")
|
|
321
|
+
out.append({
|
|
322
|
+
"adopt": None,
|
|
323
|
+
"body": "\n".join(body_parts).strip() or None,
|
|
324
|
+
"category": "guide",
|
|
325
|
+
"description": (f"Chapter {i} of {len(chapters) - 1} -- "
|
|
326
|
+
f"{ch.get('blurb', '')}\n\nAbout {int(ch.get('minutes', 0))} "
|
|
327
|
+
f"minutes. Read online: https://aitherium.github.io/awknowledge/"
|
|
328
|
+
f"path/{cid}.html"),
|
|
329
|
+
"see_also": list(ch.get("bricks") or []) + (
|
|
330
|
+
[f"guide-{i + 1:02d}"] if i + 1 < len(chapters) else []),
|
|
331
|
+
"status": "published",
|
|
332
|
+
"synopsis": str(ch.get("title", "")),
|
|
333
|
+
"topic": topic,
|
|
334
|
+
"slug": cid,
|
|
335
|
+
})
|
|
336
|
+
site = journey.get("site") or {}
|
|
337
|
+
out.append({
|
|
338
|
+
"adopt": "awkno guide 1 # start at chapter 1; `awkno open guide` for the browser",
|
|
339
|
+
"body": "\n".join(index_lines) or None,
|
|
340
|
+
"category": "guide",
|
|
341
|
+
"description": str(site.get("tagline", "The Aither World Guide")) +
|
|
342
|
+
"\n\nOnline: " +
|
|
343
|
+
str(site.get("url", "https://aitherium.github.io/awknowledge/")),
|
|
344
|
+
"see_also": [p["topic"] for p in out],
|
|
345
|
+
"status": "published",
|
|
346
|
+
"synopsis": str(site.get("title", "The Aither World Guide")),
|
|
347
|
+
"topic": "guide",
|
|
348
|
+
"slug": None,
|
|
349
|
+
})
|
|
350
|
+
return out
|
|
351
|
+
|
|
352
|
+
def _make_law_page(self, law_file: Path) -> dict[str, Any]:
|
|
353
|
+
"""Create a page for a law.
|
|
354
|
+
|
|
355
|
+
Args:
|
|
356
|
+
law_file: Path to a law markdown file.
|
|
357
|
+
|
|
358
|
+
Returns:
|
|
359
|
+
A page dict ready for JSON serialization.
|
|
360
|
+
"""
|
|
361
|
+
# Extract law number and slug from filename: 01-rule-name.md
|
|
362
|
+
stem = law_file.stem
|
|
363
|
+
parts = stem.split("-", 1)
|
|
364
|
+
law_num = int(parts[0])
|
|
365
|
+
law_slug = parts[1] if len(parts) > 1 else ""
|
|
366
|
+
|
|
367
|
+
# Read the file, and SCRUB IT — the law bodies are the longest prose in
|
|
368
|
+
# this corpus and the only source that cites internal gate scripts by
|
|
369
|
+
# name. Brick and stack pages were scrubbed from the start; this call
|
|
370
|
+
# site was missed, so a correct sanitiser shipped internal identifiers
|
|
371
|
+
# from three law pages while every other page was clean.
|
|
372
|
+
content = self._scrub_internal_refs(law_file.read_text(encoding="utf-8"))
|
|
373
|
+
|
|
374
|
+
# Extract the title (first line, usually # heading)
|
|
375
|
+
lines = content.strip().split("\n")
|
|
376
|
+
title = ""
|
|
377
|
+
description = ""
|
|
378
|
+
|
|
379
|
+
for i, line in enumerate(lines):
|
|
380
|
+
if line.startswith("# "):
|
|
381
|
+
title = line[2:].strip()
|
|
382
|
+
description = "\n".join(lines[i + 1 :]).strip()
|
|
383
|
+
break
|
|
384
|
+
|
|
385
|
+
# Use the title as the synopsis
|
|
386
|
+
synopsis = title if title else law_slug.replace("-", " ").title()
|
|
387
|
+
|
|
388
|
+
# Create topic key as "law-01" for law 1
|
|
389
|
+
topic_key = f"law-{law_num:02d}"
|
|
390
|
+
|
|
391
|
+
return {
|
|
392
|
+
"adopt": None,
|
|
393
|
+
"body": description if description else None,
|
|
394
|
+
"category": "law",
|
|
395
|
+
"description": f"Law #{law_num}",
|
|
396
|
+
"see_also": None,
|
|
397
|
+
"status": "published",
|
|
398
|
+
"synopsis": synopsis,
|
|
399
|
+
"topic": topic_key,
|
|
400
|
+
# The filename slug. `awkno law <N|SLUG>` advertised this lookup
|
|
401
|
+
# while the value was computed here and thrown away, so no slug
|
|
402
|
+
# ever resolved -- a documented invocation that could not work.
|
|
403
|
+
"slug": law_slug or None,
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def main() -> None:
|
|
408
|
+
"""CLI entry point for corpus generation."""
|
|
409
|
+
import sys
|
|
410
|
+
|
|
411
|
+
# --out DIR writes elsewhere, leaving the committed corpus untouched. Parsed by
|
|
412
|
+
# hand rather than with argparse to keep the positional repo_root working exactly
|
|
413
|
+
# as it did -- this is a published module, and a CLI that changes shape under a
|
|
414
|
+
# caller is a breakage nobody attributes to a "docs" change.
|
|
415
|
+
argv = sys.argv[1:]
|
|
416
|
+
out_dir = None
|
|
417
|
+
if "--out" in argv:
|
|
418
|
+
i = argv.index("--out")
|
|
419
|
+
if i + 1 >= len(argv):
|
|
420
|
+
print("ERROR: --out needs a directory")
|
|
421
|
+
exit(2)
|
|
422
|
+
out_dir = argv[i + 1]
|
|
423
|
+
argv = argv[:i] + argv[i + 2:]
|
|
424
|
+
|
|
425
|
+
try:
|
|
426
|
+
# Allow repo root to be passed as first argument
|
|
427
|
+
repo_root = argv[0] if argv else None
|
|
428
|
+
generator = AwknoGenerator(repo_root, output_dir=out_dir)
|
|
429
|
+
pages = generator.generate()
|
|
430
|
+
print(f"[OK] Generated corpus with {len(pages)} pages")
|
|
431
|
+
print(f" Output: {generator.output_dir}")
|
|
432
|
+
|
|
433
|
+
except FileNotFoundError as e:
|
|
434
|
+
print(f"ERROR: {e}")
|
|
435
|
+
exit(1)
|
|
436
|
+
except ValueError as e:
|
|
437
|
+
print(f"ERROR: {e}")
|
|
438
|
+
print("\nUsage: python awkno/generate.py [repo_root]")
|
|
439
|
+
print("\nIf repo_root is not provided, auto-detects from ecosystem.yaml location")
|
|
440
|
+
exit(1)
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
if __name__ == "__main__":
|
|
444
|
+
main()
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
{
|
|
2
|
+
"adopt": null,
|
|
3
|
+
"category": "stack",
|
|
4
|
+
"description": "What an agent needs to perceive things outside its own process — search and pages.\n\nIncludes: awfind, awbrowse, awdk",
|
|
5
|
+
"see_also": [
|
|
6
|
+
"awfind",
|
|
7
|
+
"awbrowse",
|
|
8
|
+
"awdk"
|
|
9
|
+
],
|
|
10
|
+
"status": "planned",
|
|
11
|
+
"synopsis": "Senses",
|
|
12
|
+
"topic": "agent-senses"
|
|
13
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"adopt": null,
|
|
3
|
+
"category": "stack",
|
|
4
|
+
"description": "awnix already carries the capabilities; adding an agent is three lines in a Dockerfile, and your skills and credentials layer on top of that. The point of the order is that the guarantees live in the base, so swapping the agent keeps them.\n\nIncludes: awnix, awdk, awskills, awkno",
|
|
5
|
+
"see_also": [
|
|
6
|
+
"awnix",
|
|
7
|
+
"awdk",
|
|
8
|
+
"awskills",
|
|
9
|
+
"awkno"
|
|
10
|
+
],
|
|
11
|
+
"status": "partial",
|
|
12
|
+
"synopsis": "The bare agent VM",
|
|
13
|
+
"topic": "agent-vm"
|
|
14
|
+
}
|