akasha-model 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. akasha_model-0.2.0/LICENSE +21 -0
  2. akasha_model-0.2.0/PKG-INFO +503 -0
  3. akasha_model-0.2.0/README.md +471 -0
  4. akasha_model-0.2.0/akasha_model/__init__.py +183 -0
  5. akasha_model-0.2.0/akasha_model/audit.py +185 -0
  6. akasha_model-0.2.0/akasha_model/authority.py +163 -0
  7. akasha_model-0.2.0/akasha_model/calibration.py +87 -0
  8. akasha_model-0.2.0/akasha_model/catalog.py +72 -0
  9. akasha_model-0.2.0/akasha_model/data.py +374 -0
  10. akasha_model-0.2.0/akasha_model/decision.py +312 -0
  11. akasha_model-0.2.0/akasha_model/ensemble.py +133 -0
  12. akasha_model-0.2.0/akasha_model/eval.py +84 -0
  13. akasha_model-0.2.0/akasha_model/gate.py +212 -0
  14. akasha_model-0.2.0/akasha_model/gate_multitask.py +278 -0
  15. akasha_model-0.2.0/akasha_model/host.py +123 -0
  16. akasha_model-0.2.0/akasha_model/model.py +147 -0
  17. akasha_model-0.2.0/akasha_model/multitask.py +422 -0
  18. akasha_model-0.2.0/akasha_model/multitask_calibrate.py +145 -0
  19. akasha_model-0.2.0/akasha_model/multitask_eval.py +285 -0
  20. akasha_model-0.2.0/akasha_model/multitask_train.py +120 -0
  21. akasha_model-0.2.0/akasha_model/outcomes.py +248 -0
  22. akasha_model-0.2.0/akasha_model/predict.py +62 -0
  23. akasha_model-0.2.0/akasha_model/primitives.py +212 -0
  24. akasha_model-0.2.0/akasha_model/rewards.py +53 -0
  25. akasha_model-0.2.0/akasha_model/rlcd.py +244 -0
  26. akasha_model-0.2.0/akasha_model/schemas/gate-audit-envelope.schema.json +44 -0
  27. akasha_model-0.2.0/akasha_model/schemas/host-outcome.schema.json +43 -0
  28. akasha_model-0.2.0/akasha_model/schemas/tool-call-plan.schema.json +44 -0
  29. akasha_model-0.2.0/akasha_model/schemas/tool-gate-request.schema.json +101 -0
  30. akasha_model-0.2.0/akasha_model/sequence.py +199 -0
  31. akasha_model-0.2.0/akasha_model/tool_calling.py +257 -0
  32. akasha_model-0.2.0/akasha_model/train.py +117 -0
  33. akasha_model-0.2.0/akasha_model/typed_decisions.py +301 -0
  34. akasha_model-0.2.0/akasha_model/vision.py +228 -0
  35. akasha_model-0.2.0/akasha_model/wire.py +281 -0
  36. akasha_model-0.2.0/akasha_model.egg-info/PKG-INFO +503 -0
  37. akasha_model-0.2.0/akasha_model.egg-info/SOURCES.txt +47 -0
  38. akasha_model-0.2.0/akasha_model.egg-info/dependency_links.txt +1 -0
  39. akasha_model-0.2.0/akasha_model.egg-info/entry_points.txt +13 -0
  40. akasha_model-0.2.0/akasha_model.egg-info/requires.txt +26 -0
  41. akasha_model-0.2.0/akasha_model.egg-info/top_level.txt +1 -0
  42. akasha_model-0.2.0/pyproject.toml +55 -0
  43. akasha_model-0.2.0/setup.cfg +4 -0
  44. akasha_model-0.2.0/tests/test_authority_profiles.py +125 -0
  45. akasha_model-0.2.0/tests/test_catalog_policy.py +85 -0
  46. akasha_model-0.2.0/tests/test_gate_budget.py +61 -0
  47. akasha_model-0.2.0/tests/test_gate_path_a.py +316 -0
  48. akasha_model-0.2.0/tests/test_smoke.py +718 -0
  49. akasha_model-0.2.0/tests/test_zeroshot_probe_smoke.py +52 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Minimal Labs
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,503 @@
1
+ Metadata-Version: 2.4
2
+ Name: akasha-model
3
+ Version: 0.2.0
4
+ Summary: One-pass typed decision model for Akasha OS: Choice, Score and Noul.
5
+ Author: Minimal Labs
6
+ License-Expression: MIT
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Provides-Extra: torch
11
+ Requires-Dist: numpy>=1.26; extra == "torch"
12
+ Requires-Dist: torch>=2.2; extra == "torch"
13
+ Provides-Extra: transformers
14
+ Requires-Dist: numpy>=1.26; extra == "transformers"
15
+ Requires-Dist: torch>=2.2; extra == "transformers"
16
+ Requires-Dist: transformers>=4.45; extra == "transformers"
17
+ Provides-Extra: eval
18
+ Requires-Dist: numpy>=1.26; extra == "eval"
19
+ Requires-Dist: torch>=2.2; extra == "eval"
20
+ Requires-Dist: datasets>=3; extra == "eval"
21
+ Provides-Extra: dev
22
+ Requires-Dist: pytest>=8; extra == "dev"
23
+ Provides-Extra: games
24
+ Requires-Dist: numpy>=1.26; extra == "games"
25
+ Requires-Dist: torch>=2.2; extra == "games"
26
+ Requires-Dist: chess>=1.11; extra == "games"
27
+ Requires-Dist: imageio>=2.37; extra == "games"
28
+ Requires-Dist: imageio-ffmpeg>=0.6; extra == "games"
29
+ Requires-Dist: pillow>=11; extra == "games"
30
+ Requires-Dist: vizdoom>=1.2.4; extra == "games"
31
+ Dynamic: license-file
32
+
33
+ # Akasha Model
34
+
35
+ Train a small one-pass model that chooses among a changing list of typed options.
36
+
37
+ Akasha Model takes a piece of state (text or JSON) and a list of questions. Each question is `choice`, `score` or `noul`. It returns one probability distribution per question in a single forward pass, instead of writing an answer word by word.
38
+
39
+ The public comparison for this shape of model is TypeSafe [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev). TypeSafe has not published its design. This repository is the decision scorer for Akasha OS: independent weights, independent training, and a deterministic tool-call planner that never executes side effects.
40
+
41
+ **Using the tool gate (authorize / abstain / block before your host runs a tool):** see [docs/using-the-tool-gate.md](docs/using-the-tool-gate.md) and [examples/gate](examples/gate/README.md). Host: `akasha_model.host` (`run_gated_call`); outcomes: `akasha_model.outcomes` (`python examples/gate/outcomes_demo.py`).
42
+
43
+ **Stable pin for OS consumers:** `pip install akasha-model==0.2.0` ([PyPI](https://pypi.org/project/akasha-model/)) or `git+https://github.com/azerothl/akasha-model.git@v0.2.0` — see [CHANGELOG.md](CHANGELOG.md). Prefer release tags over feature branches. Path A (gate/host/outcomes) installs without torch; add the `[torch]` extra for Path B scorers, training, and vision.
44
+
45
+ **Concrete use cases beyond the gate** (support triage, OS action menus, multi-signal checklists, ensemble abstention, Wikispeedia, vision lab): [docs/use-cases.md](docs/use-cases.md).
46
+
47
+ ## Demo
48
+
49
+ The same option-attention head can score controller buttons from image patches. [This ten-second film](docs/jevre-demo-10s-bgm.mp4) joins two selected five-second windows: live `deadly_corridor` combat on the seven Doom buttons, then a chess controller walking to and playing moves with five keys. The diagram shows the tensors used for each decision. The Doom window came from the supplied joint checkpoint, which averaged 0.60 kills and -97.50 reward across its ten recorded episodes. The chess window came from the stronger chess-only checkpoint, which scored 4 wins, 46 draws and 0 losses in 50 sampled games against a random mover, but 0 wins, 2 draws and 48 losses against Stockfish level 0. The windows were selected for activity and are not typical-play or competence claims.
50
+
51
+ <video src="docs/jevre-demo-10s-bgm.mp4" controls width="960"></video>
52
+
53
+ Install the game extras and record a fresh 640 by 480 Doom trace from the released joint checkpoint:
54
+
55
+ ```sh
56
+ uv pip install -e '.[games]'
57
+ python examples/doom/play.py examples/checkpoints/joint-imitation.pt --episodes 10 --game-seconds 35.3 --device cpu --capture-resolution 640x480 --output runs/doom.mp4 --trace runs/doom-trace.json
58
+ ```
59
+
60
+ Render the trace in the same visual layout. This writes a silent film because the author-owned soundtrack source is not part of the repository.
61
+
62
+ ```sh
63
+ (cd examples/film && npm install && npx playwright install chromium)
64
+ examples/film/make-film.sh runs/doom-trace.json runs/doom-film.mp4 10
65
+ ```
66
+
67
+ The release includes the [Doom example](examples/doom/README.md), the [chess example](examples/chess/README.md), the single-game checkpoints and the shared 12-option checkpoint. Both games import the visual scorer from `akasha_model.vision`; there is no second model copy in either example.
68
+
69
+ ## Architecture
70
+
71
+ The production text path encodes each question as one MASK-marker sequence:
72
+
73
+ ```
74
+ [CLS] choice question: <instructions> [SEP] [MASK] opt0 [MASK] opt1 ... [SEP] <state> [SEP]
75
+ ```
76
+
77
+ A bidirectional encoder (tiny transformer, or a **trained** BERT such as `bert-base-uncased` / ModernBERT) reads the whole sequence. A shared scorer reads the hidden state at each `[MASK]` and returns one logit per option. `choice`, `score` and `noul` share that head; noul is the two-way pair `false` / `true`.
78
+
79
+ The encoder is trained, not frozen. Options are written before the state, so a full option block can truncate the context: keep `--max-len` larger than `--head-max-len` (768 for the tiny encoder; **512 for `bert-base-uncased`**, which cannot index longer sequences). High-cardinality menus also keep `--option-max-len`; raise `--head-max-len` when a question has dozens of options.
80
+
81
+ A cheaper byte-encoder scorer remains available for CPU smoke tests and the original JSONL choice format.
82
+
83
+ ## Data format
84
+
85
+ Use one JSON object per line:
86
+
87
+ ```json
88
+ {"context":"The customer needs a refund.","options":["refund","sales","technical support"],"label":0}
89
+ ```
90
+
91
+ `label` is the zero-based index of the correct option. Each row may have a different number of options, with a minimum of two.
92
+
93
+ An option can also be an object:
94
+
95
+ ```json
96
+ {"context":"The user asks for a webpage.","options":[{"name":"tool.request","description":"Use the network tool for an external fetch."},{"name":"skill.invoke","description":"Use an installed local workflow."}],"label":0}
97
+ ```
98
+
99
+ The package exposes three typed decision contracts in `akasha_model.primitives`:
100
+
101
+ ```python
102
+ from akasha_model import ChoiceQuestion, NoulQuestion, OptionSpec, ScoreLevel, ScoreQuestion
103
+
104
+ choice = ChoiceQuestion(
105
+ "route", "Choose the safest route.",
106
+ (OptionSpec("tool.request", "Use the network tool."),
107
+ OptionSpec("skill.invoke", "Use a local skill.")),
108
+ )
109
+ score = ScoreQuestion(
110
+ "risk", "Rate the risk.",
111
+ (ScoreLevel("low"), ScoreLevel("high")),
112
+ )
113
+ noul = NoulQuestion("authorized", "Is the operation authorized?")
114
+ ```
115
+
116
+ These contracts validate bounded options, ordered score levels, probability distributions and explicit abstention thresholds. Authorization, side effects and workflow decisions remain application code rather than model output.
117
+
118
+ ### Tool calling: planning, not execution
119
+
120
+ The multitask model does not emit a JSON function call and has no tool
121
+ executor. `Choice` selects a tool from a bounded catalogue, `Score` estimates
122
+ a risk level, and `Noul` supplies binary signals such as authorization,
123
+ capability presence, or sufficient context. The application must still
124
+ validate arguments, permissions and human confirmation before any side
125
+ effect.
126
+
127
+ `akasha_model.tool_calling.ToolCallPlanner` is that deterministic boundary:
128
+ it returns a `ready`, `abstain` or `blocked` plan, validates a subset of the
129
+ parameter JSON schema, and never runs the tool. Akasha OS must supply its
130
+ own executor after a `ready` plan and repeat its final checks.
131
+
132
+ The host-facing helper is `evaluate_gate` (and `plan_scored_proposal` when a
133
+ multitask scorer fills Choice / Score / Noul). Default thresholds live in
134
+ `akasha_model.gate` (`DEFAULT_MIN_CHOICE_PROBABILITY=0.55`,
135
+ `DEFAULT_MIN_CHOICE_CONFIDENCE=0.50`, `DEFAULT_NOUL_THRESHOLD=0.70`,
136
+ `DEFAULT_MAX_RISK_SCORE=1.5`). See [examples/gate](examples/gate/README.md) for
137
+ the scripted demo, authorize-tool-call JSONL, tiny train loop, and go/no-go
138
+ checks on false positives.
139
+
140
+ ```python
141
+ from akasha_model import ToolCallPlanner, ToolSpec, evaluate_gate, ToolProposal, GateSignals
142
+
143
+ spec = ToolSpec(
144
+ "fs.read", "Read a file",
145
+ parameters={"type": "object", "required": ["path"],
146
+ "properties": {"path": {"type": "string"}},
147
+ "additionalProperties": False},
148
+ required_capability="workspace_access",
149
+ )
150
+ plan = evaluate_gate(
151
+ {spec.name: spec, "fs.delete": ToolSpec("fs.delete")},
152
+ ToolProposal("fs.read", {"path": "notes.txt"}),
153
+ GateSignals(authorized=0.95, sufficient_context=0.9, capability_present=0.92),
154
+ )
155
+ if plan.status == "ready":
156
+ akasha_executor.execute(plan.tool_name, plan.arguments)
157
+ ```
158
+
159
+ ### Multi-question training
160
+
161
+ The shared-context model accepts several atomic questions in one JSONL row.
162
+ The encoder is shared, while `Choice`, `Score` and `Noul` have independent
163
+ heads on the byte path. The MASK+BERT path scores every type at `[MASK]`
164
+ markers in one sequence.
165
+
166
+ ```json
167
+ {"context":{"request":"access a protected device","offline":true},"questions":[{"id":"route","type":"choice","instructions":"Choose the route.","options":[{"name":"deny","description":"Block the operation."},{"name":"allow","description":"Permit the operation."}],"label":0},{"id":"risk","type":"score","instructions":"Rate the risk.","levels":["low","high"],"label":1},{"id":"authorized","type":"noul","instructions":"Is it authorized?","criteria":{"true":"The user explicitly granted the capability.","false":"No explicit grant is present."},"label":0}]}
168
+ ```
169
+
170
+ Train the byte multitask scorer:
171
+
172
+ ```sh
173
+ akasha-multitask-train data/akasha_os_multi/train.jsonl \
174
+ --validation data/akasha_os_multi/validation.jsonl \
175
+ --output runs/akasha-os-multitask.pt
176
+ ```
177
+
178
+ Train the MASK encoder with RLCD (proper-scoring reward + GRPO group baseline) on the same leakage-audited splits. The tiny encoder is CPU-friendly; pass `--encoder hf --hf-model bert-base-uncased` to train BERT.
179
+
180
+ ```sh
181
+ akasha-rlcd-train data/akasha_os_multi/train.jsonl \
182
+ --validation data/akasha_os_multi/validation.jsonl \
183
+ --encoder tiny --group-size 4 \
184
+ --output runs/akasha-rlcd.pt
185
+ ```
186
+
187
+ Akasha OS rows typically hold **one `choice`, one `score` and nine `noul`**. RLCD scores every question in the row, so noul can dominate the gradient. GRPO validation loss is not accuracy. Score a MASK checkpoint with `akasha-typed-eval` (not `akasha-multitask-eval`, which only loads the byte scorer):
188
+
189
+ ```sh
190
+ akasha-typed-eval runs/akasha-rlcd.pt \
191
+ --data data/akasha_os_multi/test.jsonl \
192
+ --output reports/akasha_os_multi_rlcd.json
193
+ ```
194
+
195
+ `label` is an option index for `Choice` and a level index for `Score`. Score
196
+ levels are ordered descriptions, numbered internally from 0; they are not
197
+ arbitrary numeric measurements. A `Noul` returns the probability that the
198
+ condition in its instructions is true, with values near 0.5 representing
199
+ uncertainty. Optional `target` arrays store soft gold distributions for RLCD.
200
+
201
+ Evaluate the **byte** multitask scorer on the held-out test and stress sets:
202
+
203
+ ```powershell
204
+ akasha-multitask-eval `
205
+ runs\akasha-os-multitask.pt `
206
+ data\akasha_os_multi\test.jsonl `
207
+ --stress data\akasha_os_multi\stress.jsonl `
208
+ --output reports\akasha_os_multi_metrics.json
209
+ ```
210
+
211
+ The report includes accuracy and calibration for `Choice`, ordinal error for
212
+ `Score`, AUROC/Brier/log loss for `Noul`, coverage-risk tables, and stress
213
+ results broken down by `stress_type`.
214
+
215
+ The multi-question defaults are sized for structured Akasha states:
216
+ `context_tokens=768`, `question_tokens=512` and `option_tokens=384`.
217
+ Lower values may silently truncate criteria and source-derived descriptions.
218
+ `Noul` uses `0` or `1`. Rows may contain any subset of the three question
219
+ types; absent heads are skipped for that batch.
220
+
221
+ ## Quickstart
222
+
223
+ Run these commands from the repository root. They create local synthetic data, train on it, evaluate the saved model and score one new menu.
224
+
225
+ ```sh
226
+ uv venv
227
+ source .venv/bin/activate
228
+ uv pip install -e '.[dev,torch]'
229
+
230
+ akasha-data synthetic --output data/synthetic
231
+ akasha-train data/synthetic/train.jsonl \
232
+ --validation data/synthetic/validation.jsonl \
233
+ --output runs/synthetic.pt
234
+ akasha-eval runs/synthetic.pt data/synthetic/test.jsonl
235
+ akasha-predict runs/synthetic.pt \
236
+ --context "Choose the exact badge amber badger. Badge: amber badger." \
237
+ --option "azure crane" \
238
+ --option "amber badger" \
239
+ --option "gold heron"
240
+ ```
241
+
242
+ At inference time, option descriptions can be supplied and the application can
243
+ request abstention below a confidence threshold:
244
+
245
+ ```sh
246
+ akasha-predict runs/synthetic.pt \
247
+ --context "The user asks to fetch a public webpage." \
248
+ --option tool.request \
249
+ --option skill.invoke \
250
+ --option-description "Use the network tool for an external fetch." \
251
+ --option-description "Use an installed local workflow." \
252
+ --min-confidence 0.75
253
+ ```
254
+
255
+ The evaluation prints top-1 accuracy, which is the fraction of correct first choices. Top-3 accuracy is the fraction with the right answer among the three highest scores. It also reports negative log-likelihood (NLL), Brier score and expected calibration error (ECE). ECE compares confidence with observed accuracy. The command also prints a shuffled-context control, which pairs each menu with the wrong context. A useful model should beat that control.
256
+
257
+ ## Calibrate a trained model
258
+
259
+ Use a separate calibration split that was not used to train or select the model. Temperature scaling changes the sharpness of the probabilities without changing the ranking of the options:
260
+
261
+ ```sh
262
+ akasha-calibrate runs/synthetic.pt data/synthetic/validation.jsonl \
263
+ --output runs/synthetic-calibrated.pt
264
+ akasha-eval runs/synthetic-calibrated.pt data/synthetic/test.jsonl
265
+ ```
266
+
267
+ The fitted temperature is stored in the checkpoint and applied automatically by `akasha-predict` and `akasha-eval`. Compare NLL, Brier score and ECE on the held-out test set before deciding whether calibration helped. Calibration improves the reliability of probabilities; it does not necessarily improve top-1 accuracy.
268
+
269
+ Training can also include a calibration-aware objective. The default weight is zero, which keeps the original training behavior. A small positive weight adds the multiclass Brier score to the cross-entropy objective:
270
+
271
+ ```sh
272
+ akasha-train data/synthetic/train.jsonl \
273
+ --validation data/synthetic/validation.jsonl \
274
+ --output runs/synthetic-calibration-aware.pt \
275
+ --calibration-weight 0.1
276
+ ```
277
+
278
+ Tune this weight on a validation split. Do not assume that a lower calibration loss improves accuracy, and always compare the final held-out NLL, Brier score and ECE against the unmodified baseline.
279
+
280
+ ## Leakage-resistant splits
281
+
282
+ When related rows share the same context, split by context rather than by row. The helper below combines JSONL files and keeps every identical context in exactly one split:
283
+
284
+ ```sh
285
+ python scripts/split_by_context.py data/akasha_os data/akasha_os_grouped
286
+ ```
287
+
288
+ Train and evaluate against `data/akasha_os_grouped`. Check that the reported context sets do not overlap before interpreting the test metrics.
289
+
290
+ To probe robustness without contaminating training, derive an evaluation-only stress set from the grouped test split:
291
+
292
+ ```sh
293
+ python scripts/make_stress_eval.py \
294
+ data/akasha_os_grouped/test.jsonl \
295
+ data/akasha_os_grouped/stress.jsonl
296
+ akasha-eval runs/akasha_os_grouped/baseline.pt \
297
+ data/akasha_os_grouped/stress.jsonl
298
+ ```
299
+
300
+ The stress set contains ambiguous-context and irrelevant-noise variants. Its labels inherit the original test labels, so treat it as a robustness probe rather than a replacement for a human-labelled benchmark.
301
+
302
+ ## Ensemble abstention
303
+
304
+ For uncertainty-sensitive decisions, train several checkpoints with different seeds and evaluate their agreement:
305
+
306
+ ```sh
307
+ akasha-ensemble-eval \
308
+ runs/akasha_os_ensemble/seed-11.pt \
309
+ runs/akasha_os_ensemble/seed-23.pt \
310
+ runs/akasha_os_ensemble/seed-37.pt \
311
+ --data data/akasha_os_grouped/stress.jsonl
312
+ ```
313
+
314
+ The ensemble averages probabilities and reports the fraction of examples on which every model agrees. A conservative application can abstain when the models disagree; `unanimous_accuracy` measures the accuracy of the remaining decisions and `unanimous_coverage` measures how often a decision is still returned.
315
+
316
+ Use the same policy at inference time:
317
+
318
+ ```sh
319
+ akasha-ensemble-predict \
320
+ runs/akasha_os_ensemble/seed-11.pt \
321
+ runs/akasha_os_ensemble/seed-23.pt \
322
+ runs/akasha_os_ensemble/seed-37.pt \
323
+ --context "User asks to fetch a public webpage." \
324
+ --option tool.request \
325
+ --option skill.invoke \
326
+ --option capability.check
327
+ ```
328
+
329
+ The command abstains when the ensemble disagrees or when its mean confidence is below the configured threshold.
330
+
331
+ ## Use your own data
332
+
333
+ 1. Export train, validation and test JSONL files in the format above.
334
+ 2. Keep all options that the model will see at prediction time in each row.
335
+ 3. Split related records together. For example, keep all records for one customer or one target page in one split. This prevents near-duplicates from leaking into the test set.
336
+ 4. Run `akasha-train` (single-choice JSONL), `akasha-multitask-train` (byte Choice/Score/Noul) or `akasha-rlcd-train` (MASK+RLCD) with your train and validation files.
337
+ 5. Evaluate once on a held-out test file that was never used for training or model selection. Use `akasha-eval` for single-choice checkpoints, `akasha-multitask-eval` for the byte multitask scorer, and `akasha-typed-eval --data` for MASK/RLCD checkpoints.
338
+
339
+ The default byte encoder truncates context to 192 bytes and each option to 32 bytes. Raise `--context-tokens` or `--option-tokens` when your text needs more room. Training supports CPU, Apple MPS for a Mac GPU, and CUDA for an NVIDIA GPU through `--device`.
340
+
341
+ ### Akasha-OS-style routing data
342
+
343
+ The repository can generate a deterministic, synthetic policy dataset inspired
344
+ by [azerothl/akasha-os](https://github.com/azerothl/akasha-os): the context
345
+ describes a task, surface, resource and trust state; the options are possible
346
+ OS-level actions; and `label` identifies the action selected by a small routing
347
+ policy. It covers sessions, memory, scheduled tasks, agents, skills, tools,
348
+ model packs and capability checks. It is a local synthetic benchmark, not a
349
+ dump of Akasha OS telemetry.
350
+
351
+ ```sh
352
+ akasha-data akasha-os --output data/akasha_os
353
+ akasha-train data/akasha_os/train.jsonl \
354
+ --validation data/akasha_os/validation.jsonl \
355
+ --output runs/akasha-os.pt
356
+ akasha-eval runs/akasha-os.pt data/akasha_os/test.jsonl
357
+ ```
358
+
359
+ Splits are grouped by scenario family by default: one family (for example
360
+ Canvas, memory or device) stays entirely in a single file. For a finer
361
+ split, group by family and runtime signal instead (GPU, permissions,
362
+ network, audit, and so on):
363
+
364
+ ```sh
365
+ akasha-data akasha-os --output data/akasha_os_context --split-by context
366
+ ```
367
+
368
+ That keeps the same context situation, or a near-duplicate variant, from
369
+ appearing in both training and evaluation.
370
+
371
+ For the v2 dataset derived from a local source checkout, with abstention and
372
+ audit:
373
+
374
+ ```powershell
375
+ .venv\Scripts\python.exe scripts\generate_akasha_dataset.py `
376
+ --akasha-root <path-to-akasha-os> `
377
+ --output data\akasha_os_v2 `
378
+ --seed 20260918 `
379
+ --total 30000
380
+ .venv\Scripts\python.exe scripts\audit_akasha_dataset.py `
381
+ --input data\akasha_os_v2
382
+ ```
383
+
384
+ `stress.jsonl` is evaluation-only. The full report is in
385
+ `reports/akasha_dataset_report.md`.
386
+
387
+ ### Akasha OS — source-derived multitask dataset
388
+
389
+ To train the `Choice`, `Score` and `Noul` heads separately on scenarios
390
+ derived from identifiers that actually exist in a local Akasha OS checkout:
391
+
392
+ ```powershell
393
+ .venv\Scripts\python.exe scripts\generate_akasha_multitask_dataset.py `
394
+ --akasha-root <path-to-akasha-os> `
395
+ --output data\akasha_os_multi `
396
+ --seed 20260918 `
397
+ --total 30000
398
+ .venv\Scripts\python.exe -m akasha_model.rlcd `
399
+ data\akasha_os_multi\train.jsonl `
400
+ --validation data\akasha_os_multi\validation.jsonl `
401
+ --encoder tiny `
402
+ --output runs\akasha_os_multi_rlcd.pt
403
+ .venv\Scripts\python.exe -m akasha_model.typed_decisions `
404
+ runs\akasha_os_multi_rlcd.pt `
405
+ --data data\akasha_os_multi\test.jsonl `
406
+ --output reports\akasha_os_multi_rlcd.json
407
+ ```
408
+
409
+ This version writes 24,000 training rows, 3,000 validation rows, 3,000 test
410
+ rows and 3,000 separate `stress` rows. Each row has one Choice, one Score
411
+ and nine Noul questions. Provenance, families and the audit are in
412
+ `reports/akasha_multitask_dataset_report.md`.
413
+
414
+ ## Compare on typed-decisions
415
+
416
+ [LocalLLaMA/typed-decisions](https://huggingface.co/datasets/LocalLLaMA/typed-decisions)
417
+ is the public 2,000-decision benchmark used by Laya (0.766) and the published
418
+ Jev 1.13.0 number (0.727). Official Hub splits are 1,200 train rows and 400
419
+ test rows (2,000 test questions). They are kept as-is; this command does not
420
+ regroup rows across those splits.
421
+
422
+ A MASK checkpoint trained only on Akasha OS is a transfer run, not the
423
+ published comparison. Train on the Hub train split, then score the Hub test
424
+ split. The Hub has no validation file: use a subset of train for
425
+ `--validation` if you need early stopping, and report test only with
426
+ `akasha-typed-eval`.
427
+
428
+ ```sh
429
+ uv pip install -e '.[eval,transformers]'
430
+ akasha-typed-eval --prepare data/typed_decisions
431
+ akasha-rlcd-train data/typed_decisions/train.jsonl \
432
+ --validation data/typed_decisions/train.jsonl \
433
+ --encoder hf --hf-model bert-base-uncased \
434
+ --max-len 512 --batch-size 1 \
435
+ --output runs/akasha-typed.pt
436
+ akasha-typed-eval runs/akasha-typed.pt \
437
+ --data data/typed_decisions/test.jsonl \
438
+ --output reports/typed_decisions.json
439
+ ```
440
+
441
+ The report prints accuracy, soft accuracy, Brier, ECE and score MAE next to
442
+ those published comparison figures. A number here is agreement with a teacher
443
+ model, not a claim of correctness, and it is not a substitute for the Akasha OS
444
+ anti-leakage audit.
445
+
446
+ ## Use a trained BERT encoder
447
+
448
+ Install the optional dependency and name any MASK-capable encoder from Hugging Face. Unlike the frozen-encoder choice scorer, RLCD updates the encoder weights.
449
+
450
+ ```sh
451
+ uv pip install -e '.[transformers]'
452
+ akasha-rlcd-train data/akasha_os_multi/train.jsonl \
453
+ --validation data/akasha_os_multi/validation.jsonl \
454
+ --output runs/akasha-bert.pt \
455
+ --encoder hf \
456
+ --hf-model bert-base-uncased \
457
+ --max-len 512 \
458
+ --batch-size 1
459
+ ```
460
+
461
+ `bert-base-uncased` position embeddings stop at 512 tokens; a larger `--max-len`
462
+ warns and can index past the table. `--batch-size 4` at 768 tokens does not fit
463
+ a 16 GB GPU with group-size 4. `--device auto` picks CUDA, MPS or CPU.
464
+ `--hf-model answerdotai/ModernBERT-base` is the
465
+ closer match to Laya's published checkpoints. The checkpoint stores the trained
466
+ encoder and scorer. Loading it needs access to the same tokenizer name.
467
+
468
+ The older frozen-encoder choice path remains: `akasha-train --encoder hf`.
469
+
470
+ ## Wikispeedia example
471
+
472
+ [`scripts/get_wikispeedia.sh`](scripts/get_wikispeedia.sh) downloads the public SNAP archives and builds next-click JSONL files. The data stay outside this repository.
473
+
474
+ ```sh
475
+ scripts/get_wikispeedia.sh
476
+ akasha-train data/wikispeedia/jsonl/train.jsonl \
477
+ --validation data/wikispeedia/jsonl/validation.jsonl \
478
+ --output runs/wikispeedia.pt
479
+ ```
480
+
481
+ Cite Robert West and Jure Leskovec, *Human Wayfinding in Information Networks*, WWW 2012. Review the source data terms on the [SNAP dataset page](https://snap.stanford.edu/data/wikispeedia.html).
482
+
483
+ ## What to expect
484
+
485
+ In the experiments that led to the original starter, the one-pass scorer reached about 98% accuracy on synthetic menus. On target-disjoint Wikispeedia next-click data, a frozen Qwen2.5-0.5B encoder plus the scorer reached 26%, against about 8% for shuffled and random-encoder controls. A small model trained from scratch on 40,000 clicks reached 29%. At eight options, one pass was about 100 times faster than a small decoder forced to write 400 tokens.
486
+
487
+ These numbers describe local experiments, not this quickstart run. They are not a claim of parity with Jev.
488
+
489
+ ## Limitations
490
+
491
+ - This is a research starter for Akasha OS, not a copy of any commercial System 1 API.
492
+ - Accuracy depends on data quality, split quality and the encoder.
493
+ - The byte encoder is cheap but weak on language meaning.
494
+ - The pretrained path may download a large model and needs more memory.
495
+ - One-pass scoring requires the complete option list before prediction.
496
+ - The speed comparison used a small local decoder rather than a large commercial model.
497
+ - High-cardinality questions still need enough `--head-max-len` tokens per option, and enough `--max-len` left for the state after the option block.
498
+ - MASK/RLCD and byte-multitask checkpoints are different files. `akasha-multitask-eval` cannot load a MASK run; use `akasha-typed-eval`.
499
+ - Noul-heavy multitask rows can leave `choice` near chance even when overall GRPO loss falls. Always report per-head accuracy.
500
+
501
+ ## License
502
+
503
+ Code is released under the [MIT License](LICENSE). Downloaded datasets and pretrained models keep their own terms.