@kontextmind/kxm 0.7.91 → 0.7.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.kxm/workflows/default.yaml +1 -1
  3. package/CHANGELOG.md +212 -0
  4. package/README.md +3 -0
  5. package/docs/README.md +3 -0
  6. package/docs/agent-skills.md +123 -60
  7. package/docs/architecture.md +5 -2
  8. package/docs/cli-reference.md +3527 -0
  9. package/docs/config-reference.md +1943 -0
  10. package/docs/configuration.md +30 -4
  11. package/docs/continuous-improvement.md +122 -10
  12. package/docs/contracts/routing.md +95 -11
  13. package/docs/harness-routing.md +616 -0
  14. package/docs/kxm-handbook.md +106 -19
  15. package/docs/templates/README.md +1 -1
  16. package/docs/test-matrix.md +12 -6
  17. package/docs/troubleshooting.md +2 -2
  18. package/examples/project/.kxm/workflows/fix.yaml +1 -1
  19. package/examples/project/.kxm/workflows/improve.yaml +1 -1
  20. package/package.json +1 -1
  21. package/plugins/kxm/.claude-plugin/plugin.json +9 -10
  22. package/plugins/kxm/README.md +238 -56
  23. package/plugins/kxm/dist/claude-hook.js +10083 -0
  24. package/plugins/kxm/dist/cli.js +2487 -1848
  25. package/plugins/kxm/dist/client.js +64 -0
  26. package/plugins/kxm/dist/core.js +102 -9
  27. package/plugins/kxm/dist/extension.js +210 -68
  28. package/plugins/kxm/dist/mcp-server.js +217 -40
  29. package/plugins/kxm/dist/runtime-supervisor.js +1628 -157
  30. package/plugins/kxm/dist/runtime.js +1874 -298
  31. package/plugins/kxm/dist/server.js +416 -82
  32. package/plugins/kxm/package.json +1 -1
  33. package/plugins/kxm/skills/hints.json +1 -1
  34. package/plugins/kxm/skills/kxm/SKILL.md +48 -24
  35. package/plugins/kxm/skills/kxm/references/protocol.md +3 -3
  36. package/plugins/kxm/skills/kxm-context-memory/SKILL.md +67 -21
  37. package/plugins/kxm/skills/kxm-definitions/SKILL.md +9 -0
  38. package/plugins/kxm/skills/kxm-harness-auth/SKILL.md +82 -16
  39. package/plugins/kxm/skills/kxm-harvest/SKILL.md +1 -1
  40. package/plugins/kxm/skills/kxm-hub-ops/SKILL.md +55 -27
  41. package/plugins/kxm/skills/kxm-insights/SKILL.md +1 -1
  42. package/plugins/kxm/skills/kxm-mind/SKILL.md +2 -2
  43. package/plugins/kxm/skills/{kxm-setup → kxm-mind-setup}/SKILL.md +4 -4
  44. package/plugins/kxm/skills/kxm-peer/SKILL.md +68 -93
  45. package/plugins/kxm/skills/kxm-project-setup/SKILL.md +156 -23
  46. package/plugins/kxm/skills/kxm-projects/SKILL.md +1 -1
  47. package/plugins/kxm/skills/kxm-protocol/SKILL.md +1 -1
  48. package/plugins/kxm/skills/kxm-query/SKILL.md +1 -1
  49. package/plugins/kxm/skills/kxm-routing-improve/SKILL.md +74 -15
  50. package/plugins/kxm/skills/kxm-runs/SKILL.md +46 -17
  51. package/plugins/kxm/skills/kxm-session/SKILL.md +64 -36
  52. package/plugins/kxm/skills/kxm-skill-lifecycle/SKILL.md +44 -15
  53. package/plugins/kxm/skills/kxm-tasks/SKILL.md +16 -4
  54. package/plugins/kxm/skills/kxm-triage/SKILL.md +1 -1
  55. package/plugins/kxm/skills/kxm-work/SKILL.md +1 -1
  56. package/plugins/kxm/skills/kxm-workflow/SKILL.md +60 -19
  57. package/plugins/kxm/src/arbiter.ts +67 -22
  58. package/plugins/kxm/src/autocomplete.ts +1 -1
  59. package/plugins/kxm/src/claude-hook.ts +192 -0
  60. package/plugins/kxm/src/cli/project.ts +11 -5
  61. package/plugins/kxm/src/cli/system.ts +85 -13
  62. package/plugins/kxm/src/cli/types.ts +4 -1
  63. package/plugins/kxm/src/cli/workflows.ts +18 -16
  64. package/plugins/kxm/src/cli.ts +23 -13
  65. package/plugins/kxm/src/client.ts +15 -4
  66. package/plugins/kxm/src/commands.ts +19 -9
  67. package/plugins/kxm/src/config.ts +42 -7
  68. package/plugins/kxm/src/context-packet.ts +14 -2
  69. package/plugins/kxm/src/context.ts +16 -5
  70. package/plugins/kxm/src/dispatch-context.ts +286 -0
  71. package/plugins/kxm/src/engine-plan.ts +40 -0
  72. package/plugins/kxm/src/engine.ts +138 -6
  73. package/plugins/kxm/src/hub-env.ts +17 -1
  74. package/plugins/kxm/src/hub.ts +92 -29
  75. package/plugins/kxm/src/improve-sources.ts +228 -0
  76. package/plugins/kxm/src/improve.ts +325 -140
  77. package/plugins/kxm/src/local-snapshot.ts +101 -42
  78. package/plugins/kxm/src/mcp-server.ts +129 -30
  79. package/plugins/kxm/src/memory.ts +43 -20
  80. package/plugins/kxm/src/project-config.ts +25 -0
  81. package/plugins/kxm/src/protocol.ts +11 -0
  82. package/plugins/kxm/src/relevance.ts +138 -0
  83. package/plugins/kxm/src/retrospective.ts +16 -10
  84. package/plugins/kxm/src/runtime-service.ts +8 -1
  85. package/plugins/kxm/src/runtime-supervisor.ts +16 -2
  86. package/plugins/kxm/src/session-token-hint.ts +17 -0
  87. package/plugins/kxm/src/suggest.ts +7 -7
  88. package/plugins/kxm/src/workflow-manager.ts +80 -78
  89. package/plugins/kxm/src/workflow.ts +202 -12
  90. package/scripts/build-runtime.mjs +7 -1
  91. package/scripts/check-generated.mjs +1 -0
  92. package/scripts/emit-codex-artifacts.mjs +1 -1
@@ -11,7 +11,7 @@
11
11
  "name": "kxm",
12
12
  "source": "./plugins/kxm",
13
13
  "description": "Durable workflows, peer agents, and kxm tui",
14
- "version": "0.7.91",
14
+ "version": "0.7.93",
15
15
  "category": "development",
16
16
  "tags": ["kxm", "multi-agent", "workflows", "mcp"]
17
17
  }
@@ -42,6 +42,6 @@ steps:
42
42
  passed:
43
43
  target: $terminal
44
44
  terminalStatus: completed
45
- failed:
45
+ implementation-failure:
46
46
  target: implement
47
47
  maxTransitions: 2
package/CHANGELOG.md CHANGED
@@ -6,6 +6,18 @@ All notable user-facing changes are documented here. The project follows [Semant
6
6
 
7
7
  ### Added
8
8
 
9
+ - **`kxm workflow add --template <name>` writes a valid first workflow.**
10
+ `implement-and-verify` (the `implementer` agent, then the project's `test` gate; a
11
+ failing gate sends the work back to `implement` at most twice), `dual-critic-review` (two
12
+ review steps run as the `coordinator` agent between them) and `spec-and-plan` (plan, then
13
+ review the plan, both as `coordinator`, read-only) are written under the workflow id you
14
+ give. Each uses only what `kxm init` creates, the `coordinator` and `implementer` agents,
15
+ the `control` repository and the `test` gate, so it loads and plans as generated.
16
+ `--template` needs a workflow id (`workflow_id_required`), cannot be combined with
17
+ `--file` or `--pick` (`workflow_add_conflict`), and names the three templates when the
18
+ name is unknown (`workflow_template_unknown`); the three refusals exit 2 and honour
19
+ `--json`.
20
+
9
21
  - **Fenced hub leases, and shared external effects that will not run without one.**
10
22
  `POST /v1/leases/:resource/acquire|renew|release` are agent-authenticated and
11
23
  project-scoped (the hub prefixes the caller's project onto the resource name). Each call
@@ -57,6 +69,127 @@ All notable user-facing changes are documented here. The project follows [Semant
57
69
 
58
70
  ### Changed
59
71
 
72
+ - **The workflow loader refuses a gate step that can never settle
73
+ (`gate_outcome_impossible`).** A gate step settles only on `passed` or
74
+ `implementation-failure` when `expect` is `pass`, and only on `passed` or `repro-missing`
75
+ when `expect` is `fail`. A step that declares an outcome it never produces (typically
76
+ `failed`) and leaves one it does produce undeclared is now a load error naming the
77
+ outcomes to declare, so `kxm init`, `kxm run` and `kxm run --dry-run` report it before a
78
+ run exists. Such a workflow used to load, and a failing gate attempt could not settle:
79
+ the run was handed off with `attempt_unsettled` and later gate steps in the project were
80
+ held with `gate_recovery_pending`. An extra outcome next to every produced one still
81
+ loads, as in the `verify` step `kxm init` writes. This repository's `default` workflow,
82
+ the example project and the unsupported-gate fixture now route gate failures on
83
+ `implementation-failure`. **Check your workflows:** a gate step that routes failures only
84
+ on `failed` no longer loads.
85
+ - **`kxm run` prints how to drive the run it created, and drive refusals say why.** The
86
+ text output's second line is
87
+ `drive it model-free: kxm runs drive <runId> --simulated --wait (or cancel: kxm runs cancel <runId>)`,
88
+ and the command's help now reads "Create a KXM run (offline-first;
89
+ `kxm runs drive <runId> --simulated` executes it model-free)" instead of saying no steps
90
+ execute until the run engine lands. The JSON result and its
91
+ `phase` are unchanged. A `run_handoff_required` refusal from `kxm runs drive` now ends
92
+ with `(handoff reason …; field …; detail …)`, each part capped at 200 characters; a run of
93
+ the `default` workflow that `kxm init` writes, for example, reports `limit_unsupported`
94
+ on `limits.maxAgentTimeMs`. Top-level help names the product KXM instead of KontextMind,
95
+ and `kxm init` text output lists each validation issue as `file: code: message`.
96
+ - **`kxm suggest` recommends only KXM command skills.** Suggested skills come from the
97
+ command skills shipped in `plugins/kxm/skills` (such as `kxm-workflow`, `kxm-runs`,
98
+ `kxm-peer` and `kxm-context-memory`), never from skills that do not ship
99
+ (`troubleshooting`, `modern-web-guidance`) or from the KontextMind knowledge-plane
100
+ skills.
101
+ - **The Claude plugin's MCP errors name the user's next step, and a session appears to
102
+ peers before its first tool call.** An unreachable hub names the URL and `kxm hub start`
103
+ or `/plugin configure kxm@kxm`; `invalid_auth` names the project token; a
104
+ `session_token_invalid` denial says to unset or replace `KXM_SESSION_TOKEN` when the token
105
+ came from the environment, or to run `kxm session token --clear` when it came from the
106
+ token file. Denials still fail closed. A second concurrent session whose agent name is
107
+ already active registers once as `<name>-<pid>` and says so on stderr. In a KXM project
108
+ with a project token and a session policy that allows `kxm_inbox` and `kxm_reply`, the
109
+ server registers right after the MCP handshake instead of at the first tool call, and it
110
+ leaves the hub when stdin closes. The server instructions point Claude at `kxm_context`
111
+ and at telling the user the next step, in under 800 characters.
112
+ - **The Claude plugin README is rewritten, and its tool table is pinned to the MCP
113
+ server.** It covers requirements (`node` on `PATH`, a hub, and the `kxm` CLI for the
114
+ operator only), installing from Claude Code or the shell, each `userConfig` option and
115
+ which token to use (this project's token, never the hub admin token), what the MCP server
116
+ and the SessionStart hook do, every published MCP tool, pushed channel mode versus pull
117
+ mode, the 0.7.1 version pin with the uninstall-and-reinstall refresh (plugin options must
118
+ be entered again), and troubleshooting for each user-directed error. A test fails when
119
+ the README's `## MCP tools` rows and the server's `tools/list` disagree in either
120
+ direction. The configuration docs now say that `kxm_await` waits at most 60 seconds.
121
+ - **The skill suite is rescoped: every command has one owning skill, and `kxm-setup` is
122
+ renamed `kxm-mind-setup` with no alias.** `skill-suite.json` declares all 29 bundled
123
+ skills (13 KXM command skills, 7 browser skills, 9 KontextMind knowledge-plane skills),
124
+ and each of the 34 registered top-level `kxm` commands is owned by exactly one of them
125
+ (`models`, `routes` and `ssh` by `kxm-harness-auth`, `tenant` by `kxm-hub-ops`, `explain`
126
+ by `kxm-context-memory`). The nine knowledge-plane skills are kept; their descriptions now
127
+ start by saying they cover only the separate `kontext` CLI and `km_` tools, so they
128
+ trigger only when the user names KontextMind. The command skills were rewritten against
129
+ the current CLI help and drop stale claims (the run engine "not landed", port 8787,
130
+ `gate validate` on YAML, `fanout --idempotency-key`). `kxm-project-setup` now walks from
131
+ `kxm init` through a trust-reviewed first workflow to a simulated, receipt-verified run,
132
+ stopping where the user reviews and commits `.kxm` changes, and `kxm session brief`
133
+ (which saves a 24-hour operator token) appears only under its operator steps. **Rename:**
134
+ anything that names the `kxm-setup` skill must name `kxm-mind-setup`.
135
+ - **Context packets rank by deterministic task relevance.** `kxm context get`,
136
+ `kxm_context` and Runtime dispatch order eligible items by nine keys: open
137
+ contradictions first, project before `_shared` defaults, items that share a word with
138
+ the task before items that do not, role kind priority, a lexical BM25 score over the
139
+ item's summary and state key, confidence, authority, recency (newest first), then id.
140
+ Scoring uses a fixed English stopword list and no model, clock or randomness, so the
141
+ same records and request give the same packet. Contradiction and project-first order
142
+ are unchanged. The token budget is filled first-fit, so one oversized item no longer
143
+ stops smaller ones from fitting, and non-current state and proposed skills no longer
144
+ consume budget. `audit.relevance` reports numbers only (`taskTokens`,
145
+ `matchedCandidates`, and a rounded score per selected item).
146
+ - **`kxm_improvement_report` returns ranked, redacted cross-run signals.** Alongside the
147
+ per-area reports, `GET /v1/improvements` returns `signals`: journal entries from the
148
+ project's runs merged by evidence class, then an error's stage, then a normalized
149
+ summary that is redacted before it becomes a key. Only errors, open contradictions,
150
+ lessons and still-proposed skill candidates count. Priority is distinct runs × severity
151
+ (3/2/1) × mean run attempts × evidence confidence; an unknown run cost counts as 1 and
152
+ is labelled `unknown`, never 0; security signals rank first. The journal and
153
+ retrospective loop covers hub webhook runs only; `kxm run` (Runtime) runs have no
154
+ journal yet.
155
+ - **Recall ranks exact phrases, then token relevance, then id, and returns a relevance
156
+ per item.** `kxm context recall` and `kxm_recall` previously returned substring matches
157
+ in id order. Items that neither contain the query nor share a word with it are still
158
+ left out, and results still carry metadata only, never summaries.
159
+ - **The hub logs task and query sizes, not their text.** `context_packet_assembled` now
160
+ records `taskChars`, `taskTokens` and `matchedCandidates`, and `context_recall` records
161
+ `queryChars` and `queryTokens`. The caller still receives its own request in the
162
+ response.
163
+ - **Engine routing records carry an ask identity and only gate-negative outcomes.** Every
164
+ `routing.attempt.recorded` record carries four engine-reserved `providerMetadata` keys,
165
+ written after the producer's so a producer cannot spoof them: `workflowId`, `askSha256`
166
+ (the same for one step and agent across runs, whatever the run was asked to do),
167
+ `objectiveSha256` (the run prompt's digest) and `stepWrites`. A producer keeps up to 28
168
+ keys of its own. `agentRole` defaults to the dispatched agent. `finalOutcome` is written
169
+ only as `blocked` (a back edge) or `failed` (a producer error, an undeclared outcome or a
170
+ failing terminal); acceptance is resolved later from the event log. Records written
171
+ before this change are not backfilled.
172
+ - **`improvement.promotionPolicy` reports review readiness and never authorizes.**
173
+ `kxm improve` now reads `improvement.*` and reports, per candidate, `readyForReview` and a
174
+ reason under the configured policy: `manual_pr` is always ready for an operator PR,
175
+ `critic_quorum` waits for two critic receipts (the CLI supplies none, so it reports not
176
+ ready), and `auto_threshold` needs `minRuns` distinct runs, `minPassRate`, and a mean
177
+ recorded cost of at least `minCostSavings` over at least one cost sample. Every policy
178
+ ends at an operator PR; the old `authorized` result is gone. Values fail closed field by
179
+ field: an unknown policy is `manual_pr`, a half-life outside (0, 3650] days is 14, and
180
+ out-of-range thresholds fall back to 10, 0.95 and 0.5.
181
+ `improvement.telemetryHalfLifeDays` orders report rows through `weightedRecurrence` and
182
+ never decides candidacy.
183
+ - **`kxm routing report` reads Runtime records by default and counts only event-log
184
+ acceptance as a Runtime pass.** Without `--file` it reads the current project's Runtime
185
+ event store and then `.kxm/logs/telemetry.jsonl` (the same sources as `kxm improve`),
186
+ and `--json` output gains `sources`. A Runtime attempt counts as a pass only when its run
187
+ completed and the step was not re-entered. The ranking code is unchanged, and the rework
188
+ column still reads `transitions`, which Runtime records do not set.
189
+ - **`kxm improve --target` is removed.** It was accepted and never applied. Passing it
190
+ is now an unknown-option error. `KXM_IMPROVE_TARGET` still labels telemetry when it is
191
+ written; no report reads that label.
192
+
60
193
  - **Hub store schema v3 → v4, external-effects ledger v1 → v2.** The hub store gains a
61
194
  `leases` table and the ledger gains `lease_resource`/`fencing_token` columns. Neither has
62
195
  a migration lane: an older file is refused at open with `runtime_schema_outdated`, and the
@@ -174,6 +307,77 @@ All notable user-facing changes are documented here. The project follows [Semant
174
307
 
175
308
  ### Fixed
176
309
 
310
+ - **The Claude plugin's SessionStart hook is one bundled, read-only, project-scoped
311
+ script.** The two shell hooks it replaces (`kxm session brief --status` and
312
+ `kxm memory brief`) exited 127 without `kxm` on `PATH`, ran whichever `kxm` was on
313
+ `PATH`, minted a 24-hour operator token and wrote `.kxm/state/session-brief.json` at
314
+ every session start, ignored the `server_url` option, and had no timeout. The new hook is
315
+ `node ${CLAUDE_PLUGIN_ROOT}/dist/claude-hook.js session-start` with a 5-second timeout.
316
+ It reads only the project Claude Code opened (no walk-up), prints nothing outside a KXM
317
+ project, writes no files, mints no token, spawns nothing and always exits 0. Its context
318
+ is at most 1,500 characters of status (hub state probed at the plugin's `server_url`, up
319
+ to three of this project's active runs, the count of open requests for this agent, and
320
+ user-directed fixes) followed by the unchanged memory brief. The plugin version stays
321
+ 0.7.1, so an existing install gets the hook only after the reinstall described in the
322
+ plugin README.
323
+ - **The Claude plugin's MCP server never authenticates with the hub admin token.** With a
324
+ blank `auth_token` it fell back to the admin token saved in `hub-env.json` and registered
325
+ the agent in a project nobody had issued it a token for. It now uses `KXM_AUTH_TOKEN` or
326
+ this project's saved project token, and with neither it refuses before contacting the
327
+ hub. The operator CLI and the Runtime supervisor resolve credentials as before.
328
+ - **`kxm workflow add` writes workflows that load.** The one-step scaffold and the three
329
+ built-in templates used `role:` where an agent step needs `agent:`, the scaffold added a
330
+ top-level `id`, and the templates' gate steps named a `verify-gate` no project defines
331
+ and routed failures on `failed`. One such file in `.kxm/workflows/` made `kxm run` fail
332
+ with `run_failed` for every workflow in the project. The scaffold is now one
333
+ `implementer` step, and the templates are the ones described under Added.
334
+
335
+ - **`kxm improve` sees the Runtime's settled attempts and flags only same-ask repeats
336
+ across runs.** It read only `.kxm/logs/telemetry.jsonl`, which no Runtime step writes, so
337
+ it never saw an agent step; and on engine records it grouped per run and scored every pass
338
+ rate 0. It now reads the current checkout's Runtime event store read-only (one query over
339
+ the events table; it never creates, writes or migrates a store) plus telemetry, dropping
340
+ a telemetry copy of an attempt the store already supplied; `--file` still reads only the
341
+ named file. Each attempt's outcome is resolved from the event log: `accepted` when the run
342
+ completed and the step was not re-entered, `reworked` when the step was entered again,
343
+ `failed` when the run failed, and undecided otherwise. Simulated attempts are excluded
344
+ and counted. Groups key on workflow, step, agent role and ask; a coded-repeat candidate
345
+ needs the same ask decided in at least 2 runs, an accepted share of at least 0.75, and a
346
+ step that writes no repository, and a passing group that misses says why
347
+ (`writes-repository` or `ask-not-repeated`). The output names every source it read, with
348
+ counts; an unreadable store exits 1 with `improve_source_unreadable` and its path.
349
+ Workflow-step candidates now propose a `kind: gate` step and a `gates.yaml` entry with a
350
+ placeholder command, and skill candidates are labelled consolidation. Candidates remain
351
+ proposals; nothing is applied.
352
+ - **Runtime-dispatched agents receive committed, pinned project memory and hash-verified
353
+ promoted skills.** The engine built each agent's context packet with no project items, so
354
+ `.kxm/memory` and promoted skills never reached a `kxm run` agent. Now, when either
355
+ exists, the Runtime delivers active memory in project or operator scope and promoted
356
+ skills whose hash verifies, selected for the agent's role and step within 4,000 tokens,
357
+ and only when those files are tracked and clean at HEAD and still match the run's pinned
358
+ memory revision. Otherwise the context is withheld with a `dispatch_context_*` gap in the
359
+ packet (never in the prompt) and the step still runs. Promoted skills render under a new
360
+ `### Active Skills` heading. No hub call is made at dispatch.
361
+ - **Journal entries accept all ten categories and stage provenance.** The shared
362
+ `kxm_workflow_record` tool (MCP, Pi and `kxm workflow record`) offered 5 of the 10
363
+ categories, required an area and dropped `stageId`. It now takes every category and an
364
+ optional `stageId`; area defaults to the stage's declared area; and the hub, not the
365
+ caller, derives the attempt: the current attempt for an active or waiting stage, the last
366
+ one consumed for a finished stage (previously always one past it). Entries the hub writes
367
+ itself (checkpoint results, signal results, wait timeout, prompt expiry, degraded-quorum
368
+ approval and premature settlement) carry the stage and attempt; the checkpoint and
369
+ premature-settlement cases are the ones under test. `kxm workflow record` gained
370
+ `--stage-id` and accepts `record <runId> <category> <summary>` when area is omitted.
371
+ - **Late journal entries and promotions refresh the exported retrospective.** A terminal
372
+ run's retrospective is re-exported when an entry is recorded or a promotion decided
373
+ afterwards. Retrospectives also count only error entries as recurring error classes and
374
+ propose up to 12 ranked error and lesson signals. A promotion now publishes its update to
375
+ the run's project rather than to the run id.
376
+ - **Context packets deliver the evidence they select.** Evidence items could be selected
377
+ and budgeted but no packet section carried them; packets now have an `evidence` section,
378
+ and the repro and implementer roles receive evidence, so the error, observation and
379
+ state-change entries they recall reach them.
380
+
177
381
  - **`kxm memory sync` no longer writes this repository's agent policy into other
178
382
  projects.** A missing `CLAUDE.md` or `GEMINI.md` used to be created from a header
179
383
  copied from KXM's own instruction files — the planner/architecture-critic role, the
@@ -185,6 +389,14 @@ All notable user-facing changes are documented here. The project follows [Semant
185
389
  none exist. The block is placed at the end of a file that has no markers yet, rather
186
390
  than before a `## Do not` heading that only this repository uses.
187
391
 
392
+ - **`kxm memory sync` refuses malformed memory markers instead of eating text.** Sync
393
+ used to replace from the first `<!-- kxm:memory:start -->` to the first
394
+ `<!-- kxm:memory:end -->` wherever they appeared, so an orphan start marker got a new
395
+ block appended and the following sync deleted everything between the orphan and that
396
+ block; an end before its start duplicated text; a second block went stale. A file
397
+ must now hold exactly one start marker followed by one end marker, or neither.
398
+ Otherwise sync exits non-zero naming the file and the problem, and writes no file.
399
+
188
400
  - **Release version surfaces cover workspace packages:** a merged PR no longer
189
401
  breaks the release pipeline. `scripts/kxm-bump-version.mjs` now writes the
190
402
  version into every package manifest under `packages/`, using the same package
package/README.md CHANGED
@@ -213,6 +213,9 @@ The hub routes messages; it does not merge contexts, choose tasks, or bypass too
213
213
  | Install, configure, and use every KXM surface | [KXM Handbook](docs/kxm-handbook.md) |
214
214
  | Complete a Pi-to-Pi or Pi-to-Claude setup | [Getting started](docs/getting-started.md) |
215
215
  | Configure the hub or an agent | [Configuration reference](docs/configuration.md) |
216
+ | Look up any `kxm` command, option, or output | [CLI reference](docs/cli-reference.md) |
217
+ | Write project, workflow, agent, role, route, or price files | [Configuration file reference](docs/config-reference.md) |
218
+ | Choose a native harness or OpenRouter for the same model | [Native harness or OpenRouter](docs/harness-routing.md) |
216
219
  | Understand components and message flow | [Architecture](docs/architecture.md) |
217
220
  | Learn about agent skills | [Agent Skills](docs/agent-skills.md) |
218
221
  | Run the hub responsibly | [Operations guide](docs/operations.md) |
package/docs/README.md CHANGED
@@ -7,6 +7,9 @@ This documentation is organized by task. Start with the guide that matches what
7
7
  | [KXM Handbook](kxm-handbook.md) | Operators, Pi users, and Claude Code users | Wiki-ready installation, configuration, and complete feature guide |
8
8
  | [Getting started](getting-started.md) | Pi and Claude Code users | Complete the first successful multi-agent exchange |
9
9
  | [Configuration](configuration.md) | Users and operators | Understand every supported setting and default |
10
+ | [CLI reference](cli-reference.md) | Operators and agent authors | Every `kxm` command and subcommand with options, JSON output, and examples |
11
+ | [Configuration file reference](config-reference.md) | Project and workflow authors | Every `.kxm` file schema field by field, with validated examples and a worked two-step project |
12
+ | [Native harness or OpenRouter](harness-routing.md) | Operators choosing models | Decide which route runs a model reachable both natively and through an aggregator, and confirm which route a config line uses |
10
13
  | [Architecture](architecture.md) | Maintainers and integrators | Learn the component boundaries and message lifecycle |
11
14
  | [Terminal components](tui-components.md) | Maintainers and integrators | The reusable panel kit behind `kxm dash` and every configuration surface |
12
15
  | [Packages and workspaces](packages.md) | Maintainers | Workspace layout, Nx targets, Bun task running, and the layer gate |
@@ -1,33 +1,45 @@
1
1
  # KXM Agent Skills
2
2
 
3
- This document describes the bundled KXM Agent Skills suite: focused skills
4
- that cover current KXM top-level command groups. The suite documents the
5
- existing CLI. It does **not** land unified YAML role/project/workflow
6
- authority, admit new writers, or replace trusted `.kxm/roster.yaml` policy.
3
+ This document describes the skills bundled in `plugins/kxm/skills`. They
4
+ document the existing CLI and MCP tools. They do **not** land unified YAML
5
+ role/project/workflow authority, admit new writers, or replace trusted
6
+ `.kxm/roster.yaml` policy. For per-command flags, output, and exit codes, see
7
+ the [KXM CLI reference](cli-reference.md); skills do not repeat it.
7
8
 
8
- ## Feature-to-Skill Matrix
9
+ The suite has three groups, all declared in `plugins/kxm/skill-suite.json`:
9
10
 
10
- | Feature Area | Skill | Commands Covered | Purpose |
11
+ - 13 KXM command skills that together own every top-level `kxm` command;
12
+ - 7 browser automation skills;
13
+ - 9 skills for KontextMind, a separate product with its own `kontext` CLI and
14
+ `km_` tools.
15
+
16
+ ## KXM command skills
17
+
18
+ Each top-level command registered in `plugins/kxm/src/cli.ts` is owned by
19
+ exactly one skill. A skill can own several commands, and the `kxm` router owns
20
+ none.
21
+
22
+ | Feature area | Skill | Commands owned | Use it to |
11
23
  |---|---|---|---|
12
- | Core Routing | `kxm` | — | Select the right suite skill; state universal safety rules and portable CLI convention |
13
- | Project Setup | `kxm-project-setup` | `init`, `trust`, `config`, `completion` | Initialize, review permission changes, configure, and install shell completion |
14
- | Harness & Auth | `kxm-harness-auth` | `harness`, `auth`, `update`, `runtime`, `agent` | Inspect authenticated harness capability and operate supported runtimes/workers |
15
- | Hub Operations | `kxm-hub-ops` | `hub`, `backup`, `restore` | Run and protect the local hub and its durable SQLite state |
16
- | Session Management | `kxm-session` | `session`, `dash`, `studio` | Resume/inspect operator work and use UI capabilities each harness supports |
17
- | Peer Communication | `kxm-peer` | `peer` | Discover, send, poll/await, cancel, fan out, inbox, and reply safely |
18
- | Workflow Management | `kxm-workflow` | `workflow`, `gate` | Operate webhook workflows, waits/signals, evidence checkpoints, provenance |
19
- | Definitions | `kxm-definitions` | `role` | Manage role YAML through configuration commands; role edits do not grant trusted writer admission |
20
- | Run Management | `kxm-runs` | `run`, `runs` | Create and inspect local KXM runs while preserving execution boundaries |
21
- | Context & Memory | `kxm-context-memory` | `context`, `memory` | Query role-aware context and manage Git-authored memory proposals |
22
- | Skill Lifecycle | `kxm-skill-lifecycle` | `skills` | Govern candidate/evaluate/promote/reject/verify lifecycle |
23
- | Routing & Improve | `kxm-routing-improve` | `routing`, `improve` | Inspect real route quality/cost and propose reviewed improvements |
24
- | Tasks & Goals | `kxm-tasks` | `suggest`, `goal`, `task` | Recommend workflows and manage goals/tasks with SCM/tracker boundaries |
25
-
26
- ## Browser Automation Skills
27
-
28
- KXM includes dedicated skills for remote browser automation on self-hosted Steel (DOKS), exploratory navigation via `agent-browser`, testing with `Playwright`, and visual feedback. See [Browser Automation Guide](browser-automation.md) and [ADR-0002](adr/ADR-0002-browser-automation-steel-doks.md).
29
-
30
- | Feature Area | Skill | Purpose |
24
+ | Router | `kxm` | none | Pick the right skill or kxm_* tool and follow the universal safety rules |
25
+ | Project setup | `kxm-project-setup` | `init`, `trust`, `config`, `completion` | Set up a repository, review permission changes, and run a first workflow |
26
+ | Harness and auth | `kxm-harness-auth` | `harness`, `auth`, `update`, `runtime`, `agent`, `models`, `routes`, `ssh` | Check harness auth, update, run the Runtime supervisor, refresh models, admit routes, use SSH and Pi workers |
27
+ | Hub operations | `kxm-hub-ops` | `hub`, `backup`, `restore`, `tenant` | Inspect and bind the hub, read tenant status, back up and restore |
28
+ | Session | `kxm-session` | `session`, `dash`, `studio` | Read session and hub status and open dashboard or studio screens |
29
+ | Peer communication | `kxm-peer` | `peer` | Delegate to, fan out to, await, and answer other agents |
30
+ | Workflows and gates | `kxm-workflow` | `workflow`, `gate` | Record journal entries, pass checkpoints, and wait on signed callbacks |
31
+ | Definitions | `kxm-definitions` | `role` | Inspect or edit roles, role hosts, and model rosters without granting writer admission |
32
+ | Runs | `kxm-runs` | `run`, `runs` | Create, drive, inspect, and cancel runs, or smoke-test a workflow model-free |
33
+ | Context and memory | `kxm-context-memory` | `context`, `memory`, `explain` | Recall what the project knows, explain a context footprint, and record memory candidates |
34
+ | Skill lifecycle | `kxm-skill-lifecycle` | `skills` | Turn a repeated practice into a governed skill candidate |
35
+ | Self-improvement | `kxm-routing-improve` | `routing`, `improve` | Find what KXM learned and what repeats, and read recorded route spend |
36
+ | Tasks and goals | `kxm-tasks` | `suggest`, `goal`, `task` | Pick a workflow and plan work as goals and tasks |
37
+
38
+ ## Browser automation skills
39
+
40
+ KXM includes dedicated skills for remote browser automation on self-hosted Steel (DOKS), exploratory navigation via `agent-browser`, testing with `Playwright`, and visual feedback. They own no `kxm` command. See [Browser Automation Guide](browser-automation.md) and [ADR-0002](adr/ADR-0002-browser-automation-steel-doks.md).
41
+
42
+ | Feature area | Skill | Purpose |
31
43
  |---|---|---|
32
44
  | Browser Sessions | `kxm-browser-session` | Start, attach, inspect, and release Steel sessions on DOKS |
33
45
  | Human Takeover | `kxm-browser-takeover` | Handoff protocol for MFA, login, CAPTCHA, and sensitive consent |
@@ -37,54 +49,95 @@ KXM includes dedicated skills for remote browser automation on self-hosted Steel
37
49
  | Diagnostics & Recovery | `kxm-browser-diagnostics` | Investigate Steel connectivity, CDP errors, timeouts, and orphan cleanup |
38
50
  | Section Annotation | `kxm-browser-annotate` | Capture DOM sections, attach structured annotations, and feed changes to agents |
39
51
 
40
- ## Installation and Discovery
52
+ ## KontextMind knowledge plane (separate product)
53
+
54
+ These skills teach KontextMind: its `kontext` CLI, its server, and its `km_`
55
+ tools. They are not KXM and do not use the plugin's kxm_* tools. Each
56
+ description starts with "KontextMind knowledge plane only, the separate
57
+ kontext CLI and km_ tools, not KXM" and ends with "Use only when the user names
58
+ KontextMind, the kontext CLI, or a km_ tool", so a KXM request never loads
59
+ them. They own no `kxm` command. `plugins/kxm/skills/SUITE.md` introduces
60
+ them, and `plugins/kxm/skills/hints.json` holds their slash hints.
61
+
62
+ | Skill | Purpose |
63
+ |---|---|
64
+ | `kxm-mind` | Route a KontextMind request to the matching knowledge-plane skill |
65
+ | `kxm-query` | Search and read a mind with provenance |
66
+ | `kxm-harvest` | Draft redacted session learnings into a mind |
67
+ | `kxm-triage` | Work the KontextMind review queue |
68
+ | `kxm-work` | Read and update tracker work state and handoffs |
69
+ | `kxm-insights` | List and dismiss loop, gap, and recommendation insights |
70
+ | `kxm-projects` | Manage mind repositories and members |
71
+ | `kxm-protocol` | Explain KontextMind contracts, trailers, trust modes, and authorization |
72
+ | `kxm-mind-setup` | Connect a machine to a KontextMind server with the `kontext` CLI |
73
+
74
+ `kxm-mind-setup` was named `kxm-setup` before; the old name has no alias.
75
+
76
+ ## Installation and discovery
77
+
78
+ ### Claude Code
41
79
 
42
- ### For Pi Users
80
+ Claude Code loads `plugins/kxm/skills` directly; invoke a skill as
81
+ `/kxm:<name>`, for example `/kxm:kxm-project-setup`. The plugin also serves
82
+ the kxm_* MCP tools that the skills name.
43
83
 
44
- Pi automatically discovers skills in the `pi.skills` section of `package.json`. The KXM skills are included in the standard distribution:
84
+ ### Pi
85
+
86
+ Pi discovers the skills from the `pi` section of the root `package.json`:
45
87
 
46
88
  ```json
47
89
  {
48
- "pi.skills": [
49
- "./plugins/kxm/skills"
50
- ]
90
+ "pi": {
91
+ "skills": ["./plugins/kxm/skills"]
92
+ }
51
93
  }
52
94
  ```
53
95
 
54
- ### For Claude Users
55
-
56
- Claude plugins package the authored skills from `plugins/kxm/skills/` during the build process.
57
-
58
- ### For Other Harnesses
59
-
60
- Skills in `.agents/skills/` follow the standard agent skill format. Discovery
61
- outside Pi and Claude remains harness-specific; do not assume every consumer
62
- loads this mirror.
96
+ ### Codex and other harnesses
63
97
 
64
- ## Progressive Disclosure Usage
98
+ `scripts/emit-codex-artifacts.mjs` mirrors every declared skill into
99
+ `.agents/skills/`, which Codex reads together with the AGENTS.md command
100
+ block. Discovery by other `.agents/skills` consumers remains harness-specific.
65
101
 
66
- 1. Start with the core `kxm` skill to choose a specialized skill.
67
- 2. Use the named skill for that command group.
68
- 3. Teach only verbs and options that exist in `kxm <group> --help`.
102
+ ## Progressive disclosure
69
103
 
70
- ### Example Usage Patterns
104
+ 1. Start with the `kxm` router to choose a skill.
105
+ 2. Use the named skill for that request.
106
+ 3. Teach only verbs and options that `kxm <group> <verb> --help` prints, and
107
+ confirm a verb by its `Usage:` line.
71
108
 
72
109
  ```bash
73
110
  kxm peer list --json
74
111
  kxm peer send --target alice --content "Please review" --json
75
- kxm workflow list --json
112
+ kxm runs list --json
76
113
  kxm workflow checkpoint run_123 stage_a passed "Completed stage A" --json
77
114
  ```
78
115
 
79
- ## Verified vs Unverified Harness Limits
116
+ ## Writing skill descriptions
117
+
118
+ Skills are model-visible, and the description decides when a harness loads
119
+ one.
120
+
121
+ - Say what the skill does and when to use it, key use case first, in at most
122
+ 1024 characters.
123
+ - Write a single line that parses as strict YAML. An unquoted value cannot
124
+ contain `': '` anywhere; the strict-YAML test in
125
+ `test/core/skill-suite.test.ts` enforces this for every `SKILL.md`.
126
+ - Do not add `allowed-tools`.
127
+ - Do not use the retired Mesh product name, `pi-extensions`, or `mcp__`.
128
+ - Do not teach `--issue` on token commands, and do not make
129
+ `kxm session brief` an agent step: it saves a 24-hour operator session
130
+ token. List it only under an `## Operator steps` heading.
131
+
132
+ ## Verified vs unverified harness limits
80
133
 
81
- ### Verified Harnesses
134
+ ### Verified harnesses
82
135
 
136
+ - **Claude Code**: Loads `plugins/kxm/skills` from the installed plugin
83
137
  - **Pi**: Discovers `plugins/kxm/skills`
84
- - **Claude**: Plugin packaging of the authored skills
85
138
  - **Codex**: Consumes the generated `.agents/skills` mirror and AGENTS command block
86
139
 
87
- ### Unverified Harnesses
140
+ ### Unverified harnesses
88
141
 
89
142
  The following harnesses have discovery claims that remain explicitly unverified:
90
143
 
@@ -93,13 +146,15 @@ The following harnesses have discovery claims that remain explicitly unverified:
93
146
  - **OpenCode**: Compatibility unverified
94
147
  - **Other `.agents/skills` consumers**: Capabilities unverified
95
148
 
96
- ## Distinguishing Operational vs Governed Skills
149
+ ## Bundled skills vs governed skills
97
150
 
98
- ### Bundled Operational Skills
151
+ ### Bundled skills
99
152
 
100
- The skills in this suite (`kxm-*`) are bundled and operational by default. They map 1:1 with KXM's top-level command groups and are maintained as part of the core KXM distribution.
153
+ The skills in this suite ship with the plugin and are maintained in Git as
154
+ part of the KXM distribution. Command skills are grouped by task, so one skill
155
+ can own several command groups.
101
156
 
102
- ### Governed Candidates
157
+ ### Governed candidates
103
158
 
104
159
  Separately, `kxm skills` manages community or experimental candidates through
105
160
  create/evaluate/promote/reject/verify. Those governed skills are distinct from
@@ -108,20 +163,22 @@ this bundled suite. Telemetry cannot auto-promote a skill. See
108
163
  [Repository work delivery](skills/repo-work-delivery.md) for converting a
109
164
  repository request into a delivery prompt.
110
165
 
111
- ## Development and Maintenance
166
+ ## Development and maintenance
112
167
 
113
- ### Authoring Location
168
+ ### Authoring location
114
169
 
115
170
  Skills are authored in `plugins/kxm/skills/` as the primary source of truth.
171
+ Every skill directory must be declared in `plugins/kxm/skill-suite.json`
172
+ with `name`, `path`, `ownedCommands`, and a 10-200 character `intent`.
116
173
 
117
- ### Generated Mirror
174
+ ### Generated mirror
118
175
 
119
176
  `scripts/emit-codex-artifacts.mjs` copies owned skills byte-for-byte to
120
177
  `.agents/skills/` and leaves unrelated skills in that tree untouched. The
121
178
  suite manifest `plugins/kxm/skill-suite.json` is required; missing, symlink,
122
179
  or malformed manifests fail closed.
123
180
 
124
- ### Build Process
181
+ ### Build process
125
182
 
126
183
  1. Validate `plugins/kxm/skill-suite.json`
127
184
  2. Replace owned generated skill directories
@@ -130,6 +187,12 @@ or malformed manifests fail closed.
130
187
 
131
188
  ### Testing
132
189
 
133
- - `test/core/skill-suite.test.ts` validates command coverage and mirrors
134
- - `test/core/commands-policy.test.ts` checks taught verbs against `cli.ts`
190
+ - `test/core/skill-suite.test.ts` checks that every registered top-level
191
+ command has exactly one owner, every skill directory is declared, every
192
+ `SKILL.md` frontmatter parses as strict YAML, the knowledge-plane skills
193
+ stay separate from the command skills, and the mirrors match
194
+ - `test/core/commands-policy.test.ts` checks taught verbs against `cli.ts` and
195
+ keeps `kxm session brief` under `## Operator steps`
135
196
  - `test/core/generated-artifacts.test.ts` checks manifest-backed generated paths
197
+ - `test/core/docs-copy.test.ts` scans every Markdown file under
198
+ `plugins/kxm/skills` for removed product names
@@ -93,7 +93,7 @@ The hub validates and authenticates requests, stores agents and messages, pushes
93
93
 
94
94
  **Source of truth.** Semantics are defined by the protocol and schema types (`src/protocol.ts`, `src/workflow.ts`), the hub's durable state (`.kxm/state/kxm.db`: agents, messages, workflow runs, journal), and reviewed workspace configuration in git (`.kxm/config`). The `kxm` CLI, the Pi extension, and the Claude MCP server are **clients** of that state. When a client's behaviour differs from the hub's or a definition's contract, the contract is authoritative and the client is the defect. One deliberate locality limitation remains: `kxm workflow list` / `get` read the local SQLite file rather than the configured hub, so they only describe runs when the operator is on the hub host. Start, signal, and GitHub watch now resolve credentials from the selected active definition and use the start secret as the documented callback fallback.
95
95
 
96
- Signed webhook workflows add a durable run and coordinator message in one request. The stable provider delivery ID prevents duplicate Jira or GitHub retries. Ordered checkpoints enforce attempt limits and exact keyed evidence requirements. Local evidence can be accumulated when a coordinator enters a durable `waiting` state; a separately signed and deduplicated external result must complete the remaining named requirements before it can advance the stage. A separate journal preserves plans, decisions, contradictions, errors, and lessons for reviewed continuous improvement.
96
+ Signed webhook workflows add a durable run and coordinator message in one request. The stable provider delivery ID prevents duplicate Jira or GitHub retries. Ordered checkpoints enforce attempt limits and exact keyed evidence requirements. Local evidence can be accumulated when a coordinator enters a durable `waiting` state; a separately signed and deduplicated external result must complete the remaining named requirements before it can advance the stage. A separate journal preserves plans, decisions, contradictions, errors, lessons, and the other journal categories, each optionally bound to its stage and attempt, for reviewed continuous improvement.
97
97
 
98
98
  Peer-policy requirements add an evidence plane beside caller-authored strings.
99
99
  At run creation, configured eligible agent selectors resolve to stable producer
@@ -219,7 +219,8 @@ workflow state / journal / provenance → context engine
219
219
  ├─ episodes (journal-derived learning records)
220
220
  ├─ knowledge wiki (compiled, source-linked view)
221
221
  ├─ skill lifecycle (candidates → protected eval → promote/quarantine)
222
- └─ role-aware arbiter (per-role packets under token budgets)
222
+ └─ role-aware arbiter (task-relevance ranked, per-role packets under token budgets;
223
+ also feeds Runtime dispatch with committed, pinned memory and promoted skills)
223
224
  ```
224
225
 
225
226
  Key invariants:
@@ -227,6 +228,8 @@ Key invariants:
227
228
  - **Workflow state remains authoritative.** Journal entries are evidence, not policy.
228
229
  - **Authority never increases through derivation.** A deterministic grant floor per origin (human/workflow → policy, git → instruction, peer/tool/external/derived → evidence) is enforced at parse time.
229
230
  - **Project isolation.** Every context request is project-scoped; cross-project content fails closed.
231
+ - **Deterministic selection.** The arbiter ranks by lexical task relevance (BM25 with a fixed stopword list) inside fixed structural keys; no model, clock or randomness decides what a packet holds.
232
+ - **Committed context only at dispatch.** A Runtime-dispatched agent receives project memory and promoted skills only when they are committed, clean, and match the run's pinned memory revision; otherwise they are withheld with a gap and dispatch continues without them. Dispatch reads no hub source.
230
233
  - **Promotion is control-plane work.** Agents may propose state and skill candidates; only authorized, evidence-bound decisions promote them.
231
234
 
232
235
  ### Provider boundary