@kontextmind/kxm 0.7.95 → 0.7.96

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.kxm/README.md +39 -9
  3. package/CHANGELOG.md +1 -1
  4. package/README.md +147 -257
  5. package/SECURITY.md +21 -12
  6. package/docs/README.md +133 -54
  7. package/docs/adr/ADR-0002-browser-automation-steel-doks.md +24 -18
  8. package/docs/adr/ADR-0003-sqlite-only-store.md +100 -0
  9. package/docs/adr/ADR-0004-edge-identity-authentik.md +99 -0
  10. package/docs/adr/README.md +33 -0
  11. package/docs/concepts/architecture.md +262 -0
  12. package/docs/concepts/data-and-storage.md +194 -0
  13. package/docs/concepts/trust-model.md +152 -0
  14. package/docs/contracts/README.md +22 -14
  15. package/docs/contracts/effects-and-recovery.md +3 -0
  16. package/docs/contracts/migration.md +2 -2
  17. package/docs/contracts/routing.md +6 -5
  18. package/docs/contributing/assignment-runner.md +388 -0
  19. package/docs/contributing/ci-and-release.md +231 -0
  20. package/docs/contributing/development.md +362 -0
  21. package/docs/contributing/harness-routing-internals.md +192 -0
  22. package/docs/{packages.md → contributing/packages.md} +13 -15
  23. package/docs/{skills → contributing}/repo-work-delivery.md +20 -21
  24. package/docs/contributing/test-matrix.md +208 -0
  25. package/docs/{tui-components.md → contributing/tui-components.md} +30 -22
  26. package/docs/contributing/writing-docs.md +340 -0
  27. package/docs/glossary.md +471 -0
  28. package/docs/guides/agent-skills.md +137 -0
  29. package/docs/guides/browser-automation.md +160 -0
  30. package/docs/guides/context-and-memory.md +352 -0
  31. package/docs/guides/continuous-improvement.md +228 -0
  32. package/docs/guides/governed-skills.md +173 -0
  33. package/docs/guides/nous-providers.md +186 -0
  34. package/docs/guides/peer-messaging.md +304 -0
  35. package/docs/guides/pi-workers.md +219 -0
  36. package/docs/guides/provenance-gates.md +313 -0
  37. package/docs/guides/webhook-workflows.md +364 -0
  38. package/docs/kb/how-credentials-retrieved-safely.md +38 -12
  39. package/docs/kb/how-to-capture-and-annotate-section.md +15 -13
  40. package/docs/kb/how-to-connect-playwright-to-steel.md +16 -11
  41. package/docs/kb/how-to-recover-expired-session-or-orphan.md +26 -16
  42. package/docs/kb/how-to-resume-after-mfa.md +19 -11
  43. package/docs/kb/how-to-take-over-session.md +17 -13
  44. package/docs/kb/why-authentication-disappeared.md +22 -14
  45. package/docs/kb/why-automation-opened-different-browser.md +23 -14
  46. package/docs/kb/why-session-viewer-cannot-control.md +13 -12
  47. package/docs/operations/backup-and-restore.md +248 -0
  48. package/docs/operations/deploy.md +307 -0
  49. package/docs/operations/monitoring.md +209 -0
  50. package/docs/operations/runtime-sync.md +192 -0
  51. package/docs/operations/troubleshooting.md +265 -0
  52. package/docs/operations/upgrade.md +124 -0
  53. package/docs/prompts/browser-annotate-feedback.md +7 -7
  54. package/docs/prompts/browser-diagnose-recover.md +11 -10
  55. package/docs/prompts/browser-explore.md +7 -7
  56. package/docs/prompts/browser-repro-fix.md +7 -7
  57. package/docs/prompts/browser-start.md +12 -11
  58. package/docs/prompts/browser-takeover.md +8 -8
  59. package/docs/{cli-reference.md → reference/cli-reference.md} +83 -41
  60. package/docs/{config-reference.md → reference/config-reference.md} +159 -148
  61. package/docs/reference/configuration.md +299 -0
  62. package/docs/reference/harness-routing.md +508 -0
  63. package/docs/reference/http-api.md +203 -0
  64. package/docs/reference/tools.md +370 -0
  65. package/docs/{workflow-guide.md → reference/workflow-catalog.md} +92 -153
  66. package/docs/reference/workflow-definitions.md +286 -0
  67. package/docs/start/first-workflow.md +287 -0
  68. package/docs/start/install.md +146 -0
  69. package/docs/start/quickstart-claude-code.md +405 -0
  70. package/docs/start/quickstart-pi.md +213 -0
  71. package/docs/templates/README.md +78 -73
  72. package/docs/templates/adr.md +13 -13
  73. package/docs/templates/architecture.md +55 -71
  74. package/docs/templates/bug-fix.md +13 -16
  75. package/docs/templates/feature.md +14 -19
  76. package/docs/templates/handoff.md +44 -46
  77. package/docs/templates/postmortem.md +30 -43
  78. package/docs/templates/research.md +15 -20
  79. package/docs/templates/review.md +49 -50
  80. package/docs/templates/runbook.md +38 -30
  81. package/docs/templates/test-plan.md +16 -23
  82. package/docs/templates/test-report.md +14 -17
  83. package/examples/README.md +9 -5
  84. package/examples/provenance-workflow.json +1 -1
  85. package/examples/webhook-workflows/jira-development.json +59 -0
  86. package/examples/webhook-workflows/jira-issue-updated.json +12 -0
  87. package/package.json +1 -1
  88. package/packages/core/tui/README.md +1 -1
  89. package/plugins/kxm/.claude-plugin/plugin.json +1 -1
  90. package/plugins/kxm/README.md +31 -32
  91. package/plugins/kxm/dist/cli.js +5 -5
  92. package/plugins/kxm/dist/mcp-server.js +1 -1
  93. package/plugins/kxm/dist/runtime.js +1 -1
  94. package/plugins/kxm/package.json +1 -1
  95. package/plugins/kxm/skills/kxm/references/protocol.md +3 -1
  96. package/plugins/kxm/skills/kxm-browser-auth/SKILL.md +1 -1
  97. package/plugins/kxm/skills/kxm-browser-diagnostics/SKILL.md +5 -5
  98. package/plugins/kxm/skills/kxm-browser-explore/SKILL.md +2 -2
  99. package/plugins/kxm/skills/kxm-browser-session/SKILL.md +10 -13
  100. package/plugins/kxm/skills/kxm-browser-takeover/SKILL.md +1 -1
  101. package/plugins/kxm/skills/kxm-browser-verify/SKILL.md +1 -1
  102. package/plugins/kxm/skills/kxm-context-memory/SKILL.md +13 -4
  103. package/plugins/kxm/skills/kxm-hub-ops/SKILL.md +3 -1
  104. package/plugins/kxm/skills/kxm-mind-setup/SKILL.md +2 -1
  105. package/plugins/kxm/skills/kxm-project-setup/SKILL.md +31 -54
  106. package/plugins/kxm/skills/kxm-projects/SKILL.md +1 -1
  107. package/plugins/kxm/skills/kxm-protocol/SKILL.md +1 -1
  108. package/plugins/kxm/skills/kxm-routing-improve/SKILL.md +15 -7
  109. package/plugins/kxm/skills/kxm-runs/SKILL.md +11 -5
  110. package/plugins/kxm/skills/kxm-session/SKILL.md +1 -1
  111. package/plugins/kxm/skills/kxm-tasks/SKILL.md +9 -7
  112. package/plugins/kxm/skills/kxm-workflow/SKILL.md +10 -2
  113. package/plugins/kxm/src/cli/system.ts +1 -1
  114. package/plugins/kxm/src/cli.ts +3 -3
  115. package/plugins/kxm/src/init-guide-setup.ts +1 -1
  116. package/plugins/kxm/src/mcp-server.ts +1 -1
  117. package/plugins/kxm/src/modes.ts +1 -1
  118. package/schemas/README.md +1 -1
  119. package/docs/agent-communication-envelopes-and-gates.md +0 -553
  120. package/docs/agent-skills.md +0 -198
  121. package/docs/architecture.md +0 -245
  122. package/docs/assignment-runner.md +0 -264
  123. package/docs/browser-automation.md +0 -139
  124. package/docs/configuration.md +0 -437
  125. package/docs/continuous-improvement.md +0 -226
  126. package/docs/getting-started.md +0 -277
  127. package/docs/harness-routing.md +0 -616
  128. package/docs/kb/qa-authentik-authentication.md +0 -97
  129. package/docs/kb/qa-extension-install-and-hub-bootstrap.md +0 -85
  130. package/docs/kb/qa-hub-on-a-public-host.md +0 -48
  131. package/docs/kb/qa-sqlite-vs-duckdb.md +0 -35
  132. package/docs/kb/qa-what-the-hub-stores.md +0 -64
  133. package/docs/kxm-handbook.md +0 -1181
  134. package/docs/operations.md +0 -510
  135. package/docs/operator-pi-packages.md +0 -67
  136. package/docs/provenance-gates.md +0 -295
  137. package/docs/skills.md +0 -47
  138. package/docs/test-matrix.md +0 -132
  139. package/docs/troubleshooting.md +0 -322
  140. package/docs/webhook-workflows.md +0 -240
@@ -0,0 +1,192 @@
1
+ # Operate Runtime sync and leases
2
+
3
+ The [Runtime](../glossary.md#runtime) supervisor pushes a redacted summary of every run event to the [hub](../glossary.md#hub), so hub-side views such as `kxm tenant status` see runs from every machine. This page is for operators: it explains the sync loop, how to read its state, how to clear stalled or refused rows, and what the hub's fenced leases do and do not cover today.
4
+
5
+ ## Before you begin
6
+
7
+ - A KXM project (`kxm init`) with at least one Runtime run, for example from `kxm run`.
8
+ - A hub for this machine: `kxm hub bind <url>`, or `KXM_SERVER_URL` in the environment that starts the supervisor.
9
+ - A credential the hub accepts for the project's `prj_*` id (see [Which hub and credential sync uses](#which-hub-and-credential-sync-uses)).
10
+
11
+ ## How sync works
12
+
13
+ Sync is outbound only: the Runtime pushes, and the hub never calls the Runtime. The following diagram shows one event from commit to acknowledgement, and the operator step for a refused row.
14
+
15
+ ```mermaid
16
+ sequenceDiagram
17
+ participant E as Run event store
18
+ participant S as Supervisor sync tick
19
+ participant H as Hub
20
+ actor O as Operator
21
+ E->>E: Commit event and outbox row in one transaction
22
+ loop Every KXM_RUNTIME_SYNC_INTERVAL_MS (10 s)
23
+ S->>H: POST /v1/runtime/presence
24
+ S->>H: POST /v1/sync/events (pending rows in outbox order)
25
+ H-->>S: accepted, duplicate, conflict or rejected per event
26
+ S->>E: Ack accepted and duplicate rows
27
+ S->>E: Park conflict and rejected rows with the hub's code
28
+ end
29
+ Note over S,H: Hub unreachable or credential refused: rows stay pending and the tick backs off
30
+ O->>S: kxm runtime sync-retry, after fixing the hub side
31
+ S->>E: Re-queue parked rows
32
+ ```
33
+
34
+ ### What an outbox row contains
35
+
36
+ Every committed Runtime event writes one outbox row in the same SQLite transaction, so an event is never committed without its row. The row holds a new `kxm.sync-event.v1` object built from the event, never the event itself:
37
+
38
+ - Only allowlisted fields are copied. Every other field is dropped and named in `redaction.fieldsOmitted`.
39
+ - Registered secret values (the hub token and the values of `KXM_*_TOKEN`, `KXM_*_KEY`, `*_API_KEY` and `*_SECRET` environment variables) and credential shapes are replaced with `[redacted]`.
40
+ - Absolute paths become `[path]`, control characters are removed, and text is bounded.
41
+ - The default sync policy allows prompt titles only, bounded result summaries, evidence references, artifact metadata and changed-file paths. It allows no raw logs, diffs or environment values.
42
+
43
+ The prompt text itself stays in the local `run-prompts.json` sidecar next to the run store.
44
+
45
+ ### The sync tick
46
+
47
+ Every `KXM_RUNTIME_SYNC_INTERVAL_MS` (default 10 seconds, clamped to 250 ms to 60 seconds, read when the supervisor starts), the supervisor visits each project it owns:
48
+
49
+ 1. It posts `POST /v1/runtime/presence` with its Runtime id and host label. The hub stamps the heartbeat with its own clock.
50
+ 2. It posts pending rows to `POST /v1/sync/events` in outbox order, up to 32 rows or about 200 KB per request. The hub accepts at most 100 events and 256 KiB per request.
51
+ 3. It acknowledges rows the hub accepted or already held, and parks rows the hub refused for good.
52
+
53
+ Local execution never waits on sync. With no hub bound, rows simply stay pending.
54
+
55
+ ### How the hub decides
56
+
57
+ The hub accepts each event once, keyed by `{projectId, runId, sequence}`:
58
+
59
+ - The same bytes again are an idempotent `duplicate`.
60
+ - Out-of-order events are held. The per-run cursor is the gapless prefix, so a gap stays open until the missing event arrives.
61
+ - Different bytes under a used sequence, a project id already claimed by another hub project, a run homed on another Runtime, and an event pushed by a Runtime other than its home are refused and raise a `security_alert` in the hub log.
62
+ - An event that fails the sync schema is refused without an alert.
63
+
64
+ ### Which hub and credential sync uses
65
+
66
+ - **Hub:** `KXM_SERVER_URL` from the supervisor's environment, else the `kxm hub bind` binding, which is re-read on every tick.
67
+ - **Project:** the `prj_*` id in `.kxm/project.yaml`, not the package or directory name. The hub pins a project id to the first hub project that claims it.
68
+ - **Credential:** `KXM_AUTH_TOKEN` from the supervisor's environment, else the project token saved under that `prj_*` id in `hub-env.json`, else the saved admin token. To use a project token, key its `KXM_PROJECT_TOKENS` entry by the `prj_*` id.
69
+
70
+ > [!IMPORTANT]
71
+ > The supervisor keeps the environment of the command that started it. After you change `KXM_SERVER_URL` or `KXM_AUTH_TOKEN`, run `kxm runtime stop` and `kxm runtime start`.
72
+
73
+ ## Check sync status
74
+
75
+ `kxm runtime status` reads what the last tick saw from the supervisor's `GET /v1/sync/status`, a token-protected loopback endpoint:
76
+
77
+ ```bash
78
+ kxm runtime status
79
+ ```
80
+
81
+ Expected output:
82
+
83
+ ```text
84
+ runtime supervisor running: rtm_74573b9df67705d8f6596518 pid 41437 on 127.0.0.1:50724
85
+ sync prj_a17d607765e144cd95b93a8715d360b6: ok (pending 0, acked 1, refused 0)
86
+ ```
87
+
88
+ A hub that went away looks like this:
89
+
90
+ ```text
91
+ sync prj_a17d607765e144cd95b93a8715d360b6: blocked (pending 0, acked 1, refused 0) — last error: fetch failed; next attempt 2026-09-23T18:41:31.675Z
92
+ ```
93
+
94
+ | State | Meaning | Action |
95
+ |---|---|---|
96
+ | `ok` | Rows are being acknowledged | None |
97
+ | `no_hub` | No hub is bound or set; rows are kept locally by design | Bind a hub if you want hub-side views |
98
+ | `blocked` | Transport or credential failure; rows stay pending and the tick backs off, up to 5 minutes | Fix the cause; the next attempt resumes on its own |
99
+ | `refusing` | The hub durably refused rows, or answered without a result for some | Fix the hub side, then run `kxm runtime sync-retry` |
100
+
101
+ A project whose run store this release cannot open shows `blocked (its store is not readable by this build)` with the reason; see [Upgrade KXM](upgrade.md#understand-schema-changes). Right after the supervisor starts, the status can read `no project registered with this Runtime yet` until the first tick.
102
+
103
+ Add `--json` for the full record: `state`, the outbox counts with refusal codes and counts, `consecutiveFailures`, `storeReadable`, `hubUrl`, `lastCompletedAt`, `lastPushed`, `lastAcked`, `lastRefused`, `lastUnconfirmed`, `lastError` and `nextAttemptAt`.
104
+
105
+ The same changes appear once each in the Runtime log, `runtime/logs/kxm-runtime.jsonl` under the user state root:
106
+
107
+ ```text
108
+ {"level":"warn","component":"runtime","event":"runtime_sync_stalled","projectId":"prj_a17d607765e144cd95b93a8715d360b6","state":"blocked","pending":0,"acked":1,"refused":0,"reason":"fetch failed","nextAttemptAt":"2026-09-23T18:41:31.675Z"}
109
+ ```
110
+
111
+ ## Restart or replace the supervisor
112
+
113
+ The Runtime ID survives restarts. The registry holds one supervisor record. A new supervisor that finds it stopped, or finds its heartbeat older than 15 seconds or its process gone, takes the record over and keeps its `rtm_` ID, so every run's home Runtime stays valid. A supervisor whose record was taken over exits.
114
+
115
+ A live supervisor (running, with a fresh heartbeat and a live process) keeps the record: a second supervisor refuses to start with `runtime_supervisor_conflict`, and `kxm runtime start` reuses the running one and prints `already running`.
116
+
117
+ On start, and again on every tick, the supervisor reopens every project registered to its Runtime ID, so pending rows resume syncing without a new `kxm run`. A project that cannot reopen, such as a moved checkout or a store this build refuses, shows `blocked` with `storeReadable: false`, and the Runtime log records `runtime_sync_context_unavailable` once per change.
118
+
119
+ ## Clear a refused sync
120
+
121
+ A refused row leaves the pending queue with the hub's code, so it can neither block the rows behind it nor re-alert the hub on every tick. Refused rows are never deleted, and the Runtime never decides on its own that a refusal has become retryable.
122
+
123
+ | Code | Meaning | What to do |
124
+ |---|---|---|
125
+ | `sync_project_mismatch` | The hub already recorded this `prj_*` id under another hub project, usually from an older release that sent the package name | No command reassigns the claim; resolve it on the hub, then retry |
126
+ | `sync_home_runtime_mismatch` | The run is homed on a different Runtime, for example a run store copied to a second machine | Sync the run only from its home Runtime |
127
+ | `sync_runtime_mismatch` | The event names a different home Runtime than the one pushing it | Investigate the `security_alert`; do not retry blindly |
128
+ | `sync_sequence_reused` | The hub holds different bytes for the same run and sequence | Investigate the `security_alert`; a diverged copy of a run store is the usual cause |
129
+ | `sync_event_invalid` | The event failed the hub's sync schema, usually a release mismatch | Run the same release on the Runtime and the hub, then retry |
130
+ | `sync_row_too_large` | One row exceeds the hub's request size on its own | It will be refused again; keep it for diagnosis |
131
+ | `sync_row_unreadable` | The local outbox row is corrupt | Keep the store for diagnosis |
132
+
133
+ After the hub side is fixed, re-queue the parked rows from the project checkout:
134
+
135
+ ```bash
136
+ kxm runtime sync-retry --dry-run
137
+ kxm runtime sync-retry
138
+ ```
139
+
140
+ Expected output:
141
+
142
+ ```text
143
+ re-queued 3 refused outbox rows for prj_a17d607765e144cd95b93a8715d360b6
144
+ ```
145
+
146
+ The command needs a running supervisor and never starts one. It clears the backoff, so the rows go out on the next tick.
147
+
148
+ ## See synced runs on the hub
149
+
150
+ The hub's admin `GET /v1/ops/snapshot?project=<prj-id>` adds `homeRuntimes`: each Runtime with its host label, heartbeat and presence lease (`heartbeatAt` plus the hub's stale window, on the hub clock), and its runs with bounded title and status, `lastSequence` and `pendingGap`. A Runtime whose lease lapses is marked `orphaned`. That is a view state only: nothing moves the runs elsewhere.
151
+
152
+ `kxm tenant status` shows hub metadata next to the authoritative local run state, labelled by source. Hub workflow runs and Runtime runs use separate id spaces, so a comparison without a shared id is reported as unverified, never as agreement.
153
+
154
+ The hub counts sync traffic in `kxm_sync_events_accepted_total`, `kxm_sync_events_duplicate_total`, `kxm_sync_events_refused_total`, `kxm_sync_conflicts_total` and `kxm_runtime_heartbeats_total`; see [Monitor KXM](monitoring.md#scrape-metrics).
155
+
156
+ ## Use fenced leases
157
+
158
+ A [lease](../glossary.md#lease) lets one agent at a time act on a shared resource, such as a branch, and lets that resource refuse a holder whose time has run out.
159
+
160
+ ### What the hub provides
161
+
162
+ - `POST /v1/leases/<resource>/acquire`, `/renew` and `/release`, authenticated as a registered agent with its project token and agent key. The hub prefixes the caller's project onto the resource name, so two projects that name the same branch never contend.
163
+ - A time-to-live of 5 seconds to 10 minutes (default 5 minutes), decided on the hub clock inside one store transaction, so a client with a skewed clock cannot extend its own lease.
164
+ - A fencing token that starts at 1, stays the same across renewals, and increments whenever an expired lease is taken again. Present it with every write to the shared resource so a late holder is refused.
165
+ - Clear refusals: `lease_held` (another live holder), `lease_expired` (re-acquire to continue), `lease_superseded` (your token is stale), all HTTP 409, and `lease_not_found` (HTTP 404).
166
+ - Expired rows kept for the 7-day run-retention window, so the next takeover increments past their token, then purged.
167
+ - Metrics `kxm_leases_granted_total`, `kxm_leases_refused_total` and `kxm_leases_released_total`, and log events `lease_acquired`, `lease_renewed`, `lease_released`, `lease_denied` and `lease_purged`.
168
+
169
+ The `HubClient` class in `@kontextmind/kxm/client` wraps the three routes as `acquireLease`, `renewLease` and `releaseLease`.
170
+
171
+ ### What is not wired
172
+
173
+ > [!IMPORTANT]
174
+ > No `kxm` command or agent tool takes a lease, and the Runtime does not take one before an external effect such as a push, a pull request, a tracker issue or an outgoing webhook. Leases protect only the integrations that call the API themselves. The intended effect-and-recovery design is described in [Effects and recovery](../contracts/effects-and-recovery.md).
175
+
176
+ ## Troubleshooting
177
+
178
+ | Symptom | Cause | Fix |
179
+ |---|---|---|
180
+ | State stays `no_hub` although a hub runs | The supervisor has no `KXM_SERVER_URL` and the machine is not bound | Run `kxm hub bind <url>` |
181
+ | `blocked` with `401 invalid_auth` | The hub does not accept the credential for the `prj_*` id | Add a project token keyed by the `prj_*` id, or export the right token and restart the supervisor |
182
+ | `blocked` with `fetch failed` | The hub is down or the URL is wrong | Start the hub, or fix the binding |
183
+ | `refusing` right after an upgrade | The Runtime and the hub run different releases | Upgrade both, then run `kxm runtime sync-retry` |
184
+ | `runtime_not_running` from `kxm runtime sync-retry` | The supervisor is stopped | Run `kxm runtime start` first |
185
+ | Runs appear as `orphaned` on the hub | That Runtime stopped sending presence | Start it, or accept that its runs are no longer live |
186
+
187
+ ## Next steps
188
+
189
+ - Watch sync alongside the rest of the service: [Monitor KXM](monitoring.md)
190
+ - Protect the outbox and run stores: [Back up and restore KXM](backup-and-restore.md)
191
+ - The designed sync contract: [Hub synchronization contract](../contracts/synchronization.md)
192
+ - Hub routes for sync and leases: [Hub HTTP API reference](../reference/http-api.md)
@@ -0,0 +1,265 @@
1
+ # Troubleshoot KXM
2
+
3
+ This page is a reference of symptoms, causes and fixes, grouped by area. Start with the quick check, then go to the area that matches: install, hub and authentication, Claude Code, peer messaging, Pi workers, workflows, Runtime runs, context and memory, or operations.
4
+
5
+ ## Start with a quick check
6
+
7
+ Work from the smallest boundary outward:
8
+
9
+ 1. Run `kxm hub view`. Both `health` and `ready` must be `true`.
10
+ 2. Compare the hub URL, project and project token on every agent involved (`KXM_SERVER_URL`, `KXM_PROJECT`, and the token or plugin `auth_token`).
11
+ 3. Confirm every agent has a unique name within its project.
12
+ 4. List peers: `/kxm hub` in Pi, `kxm_list` in Claude Code, or `kxm peer list`.
13
+ 5. Read the hub log for `agent_registered`, `agent_stale` and `request_error` events; see [Monitor KXM](monitoring.md#read-the-logs).
14
+ 6. For a workflow, call `kxm_workflow_get` and work only on its `currentStage`, attempt and required evidence.
15
+ 7. For Runtime runs, run `kxm runtime status`.
16
+
17
+ ## Install
18
+
19
+ | Symptom | Cause | Fix |
20
+ |---|---|---|
21
+ | `kxm: command not found` | Only the Pi package or the Claude Code plugin was installed; neither installs the CLI | `npm install --global --omit=peer @kontextmind/kxm`; see [Install KXM](../start/install.md) |
22
+ | `kxm` works in one shell but not in another | npm's global bin directory is not on that shell's `PATH` | Run `kxm completion install` (bash and zsh also get a `PATH` entry), then open a new shell |
23
+ | Tab completion does nothing | Completion is not installed for this shell | `kxm completion install --shell bash`, `zsh` or `fish`; it is idempotent, and `--dry-run` previews it |
24
+ | Plugin tools fail on start | Node.js is older than 22.19 on the 22.x line, or older than 24 | Install a supported Node.js on the `PATH` the harness uses |
25
+
26
+ Install the CLI from the npm registry as shown above; [Install KXM](../start/install.md) covers every supported path. Set `KXM_SKIP_COMPLETION_PROMPT=1` to skip the completion offer after `kxm init`.
27
+
28
+ ### `pi update` fails with `couldn't find remote ref refs/heads/master`
29
+
30
+ The KXM default branch is `main`, and an older Pi checkout still tracks `master`. Remove the package and install it again with an explicit branch:
31
+
32
+ ```bash
33
+ pi remove git:github.com/kontextmind/kxm
34
+ pi install git:github.com/kontextmind/kxm@main
35
+ ```
36
+
37
+ ### `kxm harness list` says Claude Code is `not_detected` on Windows
38
+
39
+ npm installs Claude Code as `claude.cmd` and an extensionless shim, not `claude.exe` on `PATH`. `kxm harness list` retries `claude.exe`, then the package's inner `claude.exe`, then `claude.cmd`, and reports `issues: ["windows_shim"]` when only the shim answered.
40
+
41
+ If it still reports `not_detected`, confirm that `%AppData%\Roaming\npm` is on the `PATH` of the process that runs `kxm`, then check `claude --version` and `claude auth status`. Assignment dispatch (`just assign`) runs the inner `claude.exe`, and Pi's `node.exe` with `cli.js`, without a shell; it refuses unverified `.cmd` launchers. The same shim miss can affect `pi`.
42
+
43
+ ## Hub and authentication
44
+
45
+ ### The hub refuses to start
46
+
47
+ | Message | Fix |
48
+ |---|---|
49
+ | `KXM_PORT must be an integer between 0 and 65535` | Set a valid port, or unset `KXM_PORT` to use `7331` |
50
+ | `KXM_AUTH_TOKEN is required when binding beyond localhost` | Set `KXM_HOST=127.0.0.1`, or provide the admin token; see [Deploy KXM](deploy.md#choose-loopback-or-a-network-bind) |
51
+ | `KXM hub env file is malformed at <file>` or `does not use schema kxm.hub-env.v1` | Fix the JSON. Removing the file drops every saved token, so start with the full `KXM_PROJECT_TOKENS` map and update all clients |
52
+ | `KXM hub is already managed by PID <pid>` | A hub already owns this state directory; run `kxm hub stop` first |
53
+ | `EADDRINUSE` (address already in use) | Stop the other process, or choose another `KXM_PORT` and update every client's hub URL |
54
+ | `local_state_root_not_absolute` | Make `KXM_STATE_HOME` an absolute path |
55
+ | `runtime_schema_newer` | The database came from a newer release; do not delete it; upgrade KXM |
56
+ | `runtime_schema_outdated` | The database predates this release; see [Upgrade KXM](upgrade.md#understand-schema-changes) |
57
+
58
+ ### Agents fail with `invalid_auth` after a hub restart
59
+
60
+ `KXM_PROJECT_TOKENS` replaces the hub's saved project map, and the hub saves the replacement. A start with a one-project value removes every other project, so their agents are refused. Restart the hub with the full map; the merge command in [Start the hub](../start/quickstart-claude-code.md#3-start-the-hub) builds it from the saved file.
61
+
62
+ ### A hub PID claim is stale
63
+
64
+ KXM refuses to replace a live hub or worker claim, and `kxm hub stop` ignores a claim that is invalid, not running, or owned by someone else rather than guessing. A claim whose wrapper died is reclaimed on the next `kxm hub start`, which first stops an orphaned server child; `kxm hub stop` can stop such an orphan directly.
65
+
66
+ If a malformed claim remains, read the exact `.pid` JSON in the state directory and confirm its PID is not running and, for a hub, that nothing listens on the port. Then remove only that `.pid` file and its recorded `.stop` control file, and start once. Worker claim names include an identity digest, so do not substitute a similar file name. Never delete the state directory or the database to clear a claim.
67
+
68
+ ### Pi shows `hub:off`
69
+
70
+ - Confirm the hub is reachable from the Pi terminal (`kxm hub view`).
71
+ - Check the project name, and give the agent its project token rather than the admin token.
72
+ - Check whether a live agent already uses the same name in the project.
73
+ - Restart Pi after you change environment variables.
74
+ - For long-lived workers, check `KXM_WORKER_EXTENSION_PATHS` and `KXM_WORKER_SKILL_PATHS`; invalid paths fail before supervision starts. See [Run supervised Pi workers](../guides/pi-workers.md).
75
+
76
+ ### Admin routes return 401 or 503
77
+
78
+ `/metrics`, `/v1/ops/*`, state promotion and quorum degradation need the admin token. A project token never substitutes for it, even when you hold every project's token.
79
+
80
+ - **401 `invalid_auth`:** the request carried a project token or a wrong admin token. Once the hub has an admin token, this applies on loopback too.
81
+ - **503 `admin_auth_not_configured`:** the hub was started without an admin token, which `kxm hub start` never does. Stop the hub, set `KXM_AUTH_TOKEN`, keep the full `KXM_PROJECT_TOKENS` map, and restart against the same database. Give the admin token only to the operator terminal.
82
+
83
+ ### `kxm hub bind` refuses the URL
84
+
85
+ | Error | Fix |
86
+ |---|---|
87
+ | `hub_url_invalid` | Use an `http` or `https` URL without user info, query or fragment |
88
+ | `hub_bind_unauthenticated` | The URL is remote and this machine has no token for the project; export it and bind again |
89
+ | `hub_credential_unreadable` | Repair or remove `hub-env.json` under the user state root |
90
+
91
+ ## Claude Code plugin and MCP
92
+
93
+ The [plugin troubleshooting guide](../../plugins/kxm/README.md#troubleshooting) has the full list. Run fixes in your own terminal, and never paste a token into Claude.
94
+
95
+ | Symptom | Fix |
96
+ |---|---|
97
+ | `kxm_*` tools do not appear | Check `/mcp` for the `kxm` server, check `node --version`, run `/reload-plugins`, and confirm `claude plugin list` shows `kxm@kxm` enabled |
98
+ | `KXM hub unreachable at <url>` | Start the hub, or correct `server_url` with `/plugin configure kxm@kxm` |
99
+ | `no project token for project <p>` | Enter the project token at `/plugin configure kxm@kxm`, or add `<p>` to the hub's full `KXM_PROJECT_TOKENS` map and restart the hub |
100
+ | `KXM hub rejected the project token for project <p>` | Enter the token the hub holds for `<p>` |
101
+ | `tool_policy_denied: Session token on disk is malformed or expired` | Run `kxm session token --clear` |
102
+ | Pushed requests never arrive | Start Claude Code with `claude --dangerously-load-development-channels plugin:kxm@kxm` and accept the trust prompt, or use `kxm_inbox` and `kxm_reply` |
103
+ | `kxm is already at the latest version` but the plugin is old | The release job sets the version only inside its build and never commits it, so updates never refresh the plugin. [Reinstall it](../start/quickstart-claude-code.md#update-the-plugin) |
104
+ | No KXM brief at session start | Start Claude Code from the directory that contains `.kxm/` |
105
+
106
+ ## Peer messaging
107
+
108
+ | Symptom | Cause | Fix |
109
+ |---|---|---|
110
+ | An expected peer is missing | Different `KXM_PROJECT` values, or the peer stopped sending heartbeats | Compare settings and look for `agent_stale`; names display case-sensitively but live-name uniqueness ignores case |
111
+ | Registration returns HTTP 409 `duplicate_agent_name` | A live agent already uses the name in this project | Stop the old session, or choose another name |
112
+ | A request stays `queued` | The recipient has no active event stream | Confirm its process is connected; proxies must not buffer `/v1/events` |
113
+ | A request stays `delivered` | The recipient acknowledged it and is still working, waiting for approval, or blocked | Check the recipient directly; do not resend. `kxm_cancel` changes hub state only; it cannot undo work already done |
114
+ | A message disappears after completion | Terminal messages are purged after 7 days | Raise `KXM_MESSAGE_RETENTION_MS`, and keep durable results in Git |
115
+
116
+ ### `kxm_await` times out
117
+
118
+ `kxm_await` waits at most 60 seconds, which is both its default and its cap. A timeout does not end the request. Check it with `kxm_get`, and use `kxm_workflow_wait` for long external work. `cancelled`, `expired` and `error` are terminal. Resend only work that is safe to repeat, with an idempotency key after an uncertain network result.
119
+
120
+ ### Fanout returns pending before a model replies
121
+
122
+ `kxm_fanout.timeoutMs` is a local wait, not the message lifetime. A pending result includes the durable `messageId`, its status and expiry, and whether the wait timed out or was aborted. Inspect it with `kxm_get`, or repeat the exact fanout with the same correlation ID, idempotency prefix, targets and content. Do not send a replacement with a new prefix. Usually omit `ttlMs` for model work, and never count a pending peer toward a checkpoint. See [Message peer agents](../guides/peer-messaging.md).
123
+
124
+ ## Pi workers
125
+
126
+ | Symptom | Cause | Fix |
127
+ |---|---|---|
128
+ | A continued Pi session rejects every turn | The worker stopped during a tool call, leaving a `tool_use` without its `tool_result` | Nothing: the worker retries once without `--continue` and writes a recovery envelope; keep the same project and agent name |
129
+ | The agent settles on a quota or provider error | Pi's own retries are exhausted | Set `--fallback-models` (or `KXM_WORKER_FALLBACK_MODELS`) to rotate at once; otherwise the worker retries after `KXM_WORKER_PROVIDER_RETRY_MS` |
130
+ | The heartbeat is healthy but one tool never finishes | A tool exceeded `KXM_WORKER_TOOL_TIMEOUT_MS` (31 minutes by default, above the 30-minute fanout wait) | The worker logs `worker_tool_timeout` and restarts the child; raise the limit only above the longest legitimate call |
131
+ | A long-lived worker keeps restarting | Pi is missing from the service `PATH`, the working directory is gone, or model credentials are missing | Read `worker_process_error` and `worker_exited`; set `KXM_PI_COMMAND` to an explicit path |
132
+ | A read-only reviewer edits files | Prompt wording does not remove tools | Set `KXM_WORKER_TOOLS=read,grep,find,ls` |
133
+ | The worker exits at once with `pi_native_impersonation_blocked` | `--model` or a `--fallback-models` entry is a model whose vendor has its own harness, such as `xai/…`, `openai-codex/…` or `openrouter/x-ai/…` | Run that model in its native harness, or pick an admitted Pi route such as `openrouter/qwen/qwen3-coder-plus`; see [Harness routing](../reference/harness-routing.md#what-the-brake-refuses) |
134
+
135
+ Use `--fresh-start`, not `--no-continue`, when only the first launch must avoid old session state.
136
+
137
+ ### A workflow message stays queued while the worker restarts once
138
+
139
+ With `--session-isolation workflow`, a message for a different run is deliberately left queued in the current Pi context. Look for `worker_session_routed`: the old child closes, one child starts with the run's `--session-dir`, and the same message ID replays and becomes `delivered`.
140
+
141
+ If it repeats, read `worker_session_request_rejected` and check that the worker was started with `kxm agent worker`, that `KXM_WORKER_SESSION_SCOPE` was not set by hand, that only the service account can write the state directory, that the hub and worker run the same release, and that the message has a hub-owned `workflowRunId`. Never acknowledge the message by hand, edit the route request, copy a run history into `default`, or start a second worker with the same identity.
142
+
143
+ ### Session events
144
+
145
+ | Event | Meaning | Action |
146
+ |---|---|---|
147
+ | `worker_session_routed` | Expected swap to another run's session | None unless it repeats for one message |
148
+ | `worker_session_evicted` | An inactive run history was removed at the retention bound | Keep workflow facts in the journal, assets or Git |
149
+ | `worker_session_state_recovered` | A malformed binding manifest was quarantined as `.corrupt-<timestamp>` and routing restarted at `default` | Read `kxm_workflow_get`, re-drive unfinished work from durable message IDs, keep the quarantined file, and check disk and permissions |
150
+ | `worker_continue_fallback` | Pi history could not continue, so the same binding started fresh | Read the recovery envelope and the durable message state |
151
+
152
+ If a workflow seems to remember another run, confirm the worker log says `"sessionIsolation":"workflow"`. Isolation is off by default; restart with `--session-isolation workflow`. The first isolated start uses fresh storage, and history from a formerly shared session cannot be separated afterward. See [Run supervised Pi workers](../guides/pi-workers.md).
153
+
154
+ ## Workflows and gates
155
+
156
+ ### A workflow tool returns `workflow_forbidden`
157
+
158
+ Read `operation`, `assignedCoordinatorName` and `nextAction` in the error. Only the assigned coordinator can read, wait on, checkpoint or journal a run; do not retry as a peer.
159
+
160
+ ### A workflow cannot advance
161
+
162
+ A passing checkpoint needs a keyed, non-empty value for every declared `requiredEvidence` key; extra keys do not count. Warnings and failures stay active until corrected. When attempts run out, or the coordinator settles early, the run fails and its journal records why. Decide whether repeating external effects is safe before you start a new delivery.
163
+
164
+ For a `peer-reply` requirement, read `resolvedEvidencePolicies`, `verifiedEvidence` and the current attempt. An evidence string cannot satisfy it; the checkpoint must cite replied message IDs in `evidenceRefs` before those messages are purged, and every eligible agent must have registered before the run started.
165
+
166
+ | Code | Meaning |
167
+ |---|---|
168
+ | `workflow_context_forbidden` | The sender is not the run's assigned coordinator |
169
+ | `workflow_context_inactive` | The run or stage is not running |
170
+ | `workflow_context_attempt_mismatch` | Use `stage.attempts + 1`, and send fresh work after a retry |
171
+ | `workflow_evidence_producer_forbidden` | The target is not in the run's eligible producer set |
172
+ | `workflow_evidence_policy_missing`, `workflow_evidence_policy_unresolved` | The requirement has no usable peer policy |
173
+ | `workflow_provenance_invalid` | A cited message is missing, pending, ineligible, or bound to another project, run, stage, requirement or attempt |
174
+ | `workflow_evidence_incomplete` | Fewer unique verified producers than the minimum; several replies from one peer count once |
175
+
176
+ If the policy declares a lower `degradation.minProducers`, an operator with the admin token can approve it for the current stage and attempt. Preview with `kxm gate degrade <run-id> <stage-id> --requirement <key> --reason <text> --dry-run --json`, then run it without `--dry-run`. Approval does not advance the run; the coordinator still checkpoints. See [Peer provenance and quorum gates](../guides/provenance-gates.md).
177
+
178
+ ### GitHub checks passed but the run is still waiting
179
+
180
+ The hub does not poll GitHub. Run `kxm gate github watch` with the run's `--run-id`, `--stage-id` and `--signal-key`. On timeout it posts the signed `failed` signal and exits `4`; it never invents `passed`.
181
+
182
+ ### A webhook or callback is rejected
183
+
184
+ | Response | Meaning |
185
+ |---|---|
186
+ | 401 | The HMAC-SHA256 signature is missing or does not match the raw body; a callback must use `signalSecretEnv` when the definition sets it |
187
+ | 400 | The delivery ID or JSON body is missing; `workflow_evidence_incomplete` names `missingRequirements`; `invalid_workflow_evidence` means non-string or duplicate keys |
188
+ | 404 | The definition or run ID does not match this hub |
189
+ | 409 `workflow_target_unavailable` | The configured coordinator has never registered; start it once with the matching project and name |
190
+ | 409 `workflow_not_waiting` | No wait is active: `kxm_workflow_wait` was not called, the deadline failed the run, or a prior signal advanced it |
191
+ | 409 `workflow_signal_mismatch`, `workflow_signal_context_mismatch` | Use the run's exact `waiting.signalKey`; fix or omit `workflow.run`, `workflow.stage` and `workflow.signal` evidence |
192
+ | 204 | The event or filter did not match, so no run was intended |
193
+ | 200 with `duplicate: true` | A retry of the same delivery ID was deduplicated |
194
+
195
+ A failed or timed-out callback consumes that wait. Re-enter the wait and start a new `github watch` or `kxm gate signal`; keep an explicit `--delivery-id` only for retries of one unchanged body. See [Run webhook workflows](../guides/webhook-workflows.md).
196
+
197
+ ### A workflow file fails to load with `gate_outcome_impossible`
198
+
199
+ A gate step declares an outcome its `expect` value never produces. The message names the outcomes to declare instead.
200
+
201
+ ## Runtime runs
202
+
203
+ | Symptom | Cause | Fix |
204
+ |---|---|---|
205
+ | `runtime supervisor is not running` or `runtime_not_running` | The supervisor stopped | `kxm runtime start`; commands such as `kxm run` also start it on demand |
206
+ | `runtime_supervisor_unreachable` | A live supervisor process does not answer its token probe | Check the PID from `kxm runtime status`, stop a hung process with your OS tools, then `kxm runtime start` |
207
+ | `project_required` | The command ran outside a KXM project | Run it from the checkout, or run `kxm init` |
208
+ | `producer_route_not_admitted` | A live drive uses a model route that is not admitted | `kxm routes admit --model <provider/model>`, or drive with `--simulated` |
209
+ | `pi_not_authenticated: pi harness not detected (pi_native_impersonation_blocked)` | A Pi agent's model belongs to a vendor with its own harness | Set the agent's native `harness:`, or choose an admitted Pi route whose vendor has none |
210
+ | `run_busy` (HTTP 409) | The run is already admitted or queued for a drive | Wait, and check `kxm runs status <run-id>` |
211
+ | A run store is refused with `runtime_schema_outdated` | The store predates this release | See [Upgrade KXM](upgrade.md#understand-schema-changes) |
212
+ | Runs do not reach the hub | Sync is `no_hub`, `blocked` or `refusing` | See [Operate Runtime sync and leases](runtime-sync.md#check-sync-status) |
213
+
214
+ ## Context and memory
215
+
216
+ | Symptom | Cause | Fix |
217
+ |---|---|---|
218
+ | 403 `context_isolation_violation` | An agent asked for another project's context | Use the agent's own project; cross-project reads need the admin token |
219
+ | State promotion returns 401 or 503 | Promotion needs the configured admin token; agents can only propose | Promote with `kxm context promote <project> <proposal-id>` from the operator terminal |
220
+ | `kxm context recall` returns nothing | No live record matches; superseded and rejected records are excluded | Broaden the query, or check `unresolvedGaps` |
221
+ | A new memory fact does not appear in `kxm memory brief` | `kxm memory note` records a candidate, which becomes active only when promoted through a pull request | Review and merge the candidate, then run `kxm memory sync` |
222
+ | `memory sync failed: none of AGENTS.md, CLAUDE.md, GEMINI.md exists` | Sync updates only the instruction files a project already has | Create the file your harness reads, then sync again |
223
+ | `memory sync failed: <file> has …; wrote no file` | An orphan memory marker, an end marker before its start, or a second block | Keep exactly one `kxm:memory` marker pair in the file, or delete both, then sync again |
224
+
225
+ See [Context and memory](../guides/context-and-memory.md).
226
+
227
+ ## Operations
228
+
229
+ | Failure | Behavior | Recovery |
230
+ |---|---|---|
231
+ | An agent exits | It is marked offline after the stale window | Restart it with the same name to resume its ID |
232
+ | The event stream drops | The client reconnects while heartbeats continue | If it repeats, check the network and proxy buffering |
233
+ | The hub or a worker exits | SQLite keeps agents and messages, and binding manifests keep the Pi scope | Restart. Queued messages are pushed again, by the same ID, until acknowledged; delivered messages are not replayed |
234
+ | The disk fails or fills | Readiness or writes fail | Restore storage, then check database integrity and `/ready` |
235
+ | An external callback is lost | The run stays `waiting` until its deadline, then fails and notifies the coordinator | Retry with the same delivery ID, or review side effects before a new delivery |
236
+ | Requests return 429 `rate_limited` | The per-agent or per-address window is full | Honor `Retry-After`; raise `KXM_RATE_LIMIT_MAX` if the load is expected |
237
+ | `kxm restore` refuses with `runtime_schema_mismatch` | A backup file's schema version differs from the one its manifest records | Use another backup set; never mix files between sets |
238
+
239
+ For backup and restore errors such as `backup_no_stores`, see [Back up and restore KXM](backup-and-restore.md#troubleshooting). The dashboard's action keys do not act on runs; see [Monitor KXM](monitoring.md#keys).
240
+
241
+ ## Collect a useful bug report
242
+
243
+ Include:
244
+
245
+ - the operating system and Node.js version;
246
+ - the Pi or Claude Code version, and the KXM version (`kxm --version`) or Git commit;
247
+ - whether the hub is local or behind a proxy;
248
+ - redacted environment values, never tokens;
249
+ - the relevant structured hub events;
250
+ - exact reproduction steps and the expected behavior.
251
+
252
+ Never attach tokens, private prompts, credentials, raw `pi-agent-*.log` files, or unrelated repository content.
253
+
254
+ ## For maintainers
255
+
256
+ - **`kxm --help` prints an old flat command list.** The committed `plugins/kxm/dist/cli.js` is stale. Run `npm run build` and commit the generated `dist`.
257
+ - **Every CI job stays queued while a runner is online.** See [CI and release](../contributing/ci-and-release.md#ci-jobs-stay-queued-while-a-runner-is-online).
258
+ - **Loading an exact extension in Pi during development.** Use `pi --no-extensions -e ./plugins/kxm/src/extension.ts`, adding every required provider extension with another `-e`; see [Develop KXM](../contributing/development.md).
259
+
260
+ ## Related
261
+
262
+ - [Monitor KXM](monitoring.md)
263
+ - [Deploy KXM](deploy.md)
264
+ - [CLI reference](../reference/cli-reference.md)
265
+ - [Claude Code plugin](../../plugins/kxm/README.md)
@@ -0,0 +1,124 @@
1
+ # Upgrade KXM
2
+
3
+ Move the `kxm` CLI, the hub, the Runtime, and the Claude Code or Pi integrations to a new release, and roll back if the new release misbehaves. This page is for operators. KXM stores have no migrations, so the order of steps matters more than usual.
4
+
5
+ ## Before you begin
6
+
7
+ - Protected storage for a full backup; see [Back up and restore KXM](backup-and-restore.md).
8
+ - A window in which the hub, the Runtime supervisor and long-lived workers may stop.
9
+ - The GitHub CLI (`gh`, signed in) if you apply updates with `kxm update --kxm` from the default `github` source, which downloads the release tarball with it.
10
+
11
+ ## Check for an update
12
+
13
+ ```bash
14
+ kxm update --check
15
+ ```
16
+
17
+ Expected output when a newer release exists:
18
+
19
+ ```text
20
+ kxm 0.7.1 → 0.7.2 available · kxm update --kxm
21
+ ```
22
+
23
+ From a source checkout, the check makes no network call and reports the checkout instead:
24
+
25
+ ```text
26
+ kxm 0.7.1 (running from source at /work/kxm)
27
+ ```
28
+
29
+ The check reads `update.yaml` under the user state root. Its `source` is `github` (release tarballs) by default; `npm` is also accepted. See [Updater settings](../reference/config-reference.md#updater-settings-kxmupdatev1). Add `--json` to see the install kind KXM detected.
30
+
31
+ ## Stop the services and back up
32
+
33
+ Stop everything that holds a KXM database open, in this order, and confirm each one is down:
34
+
35
+ ```bash
36
+ kxm runtime stop
37
+ kxm hub stop # also stops long-lived workers that recorded PID claims
38
+ kxm runtime status # expect "runtime supervisor is not running"
39
+ ```
40
+
41
+ Pause service-manager restarts and background hub starts (`hub.autoStart`) until the upgrade finishes. An old process that keeps running against a store the new release rewrites, or a new process that opens an old store, fails closed at best.
42
+
43
+ Then take the stopped-state backup in [Back up everything else](backup-and-restore.md#back-up-everything-else), and record the current version with `kxm --version`.
44
+
45
+ ## Upgrade the CLI
46
+
47
+ Use the path that matches how KXM was installed. `kxm update --kxm` applies an update only for a global npm install; for every other install kind it exits `2` and prints the command to use instead.
48
+
49
+ | Install kind | Upgrade with |
50
+ |---|---|
51
+ | Global npm install | `kxm update --kxm` (release tarball, or npm with `source: npm`), or `npm install --global --omit=peer @kontextmind/kxm@latest` |
52
+ | Pi package | `pi update` |
53
+ | Claude Code marketplace plugin | Reinstall the plugin; see the plugin note below |
54
+ | Project dependency | `npm install @kontextmind/kxm@latest` in that project |
55
+ | Source checkout | `git pull`, then `npm ci` |
56
+
57
+ `kxm update --kxm` refuses to install a GitHub release that does not publish a SHA-256 digest for its tarball (`release_digest_missing`) or whose download does not match it (`release_digest_mismatch`). Preview the steps first:
58
+
59
+ ```bash
60
+ kxm update --kxm --dry-run
61
+ ```
62
+
63
+ ### Update harnesses and plugins
64
+
65
+ Without `--check` or a lone `--kxm`, `kxm update` also runs the native updaters of detected harnesses. Narrow it with a harness id and one scope:
66
+
67
+ ```bash
68
+ kxm update --dry-run # plan every detected harness
69
+ kxm update pi --extensions # pi update --extensions
70
+ kxm update claude --extensions # claude plugin update kxm -y (user scope); see the note below
71
+ kxm update pi --models # refresh Pi model catalogs
72
+ ```
73
+
74
+ > [!IMPORTANT]
75
+ > The release job sets the plugin's version only inside its build and never commits it, so the plugin's version never changes. Claude Code installs new plugin code only when that version changes, so `claude plugin update` (and `kxm update claude --extensions`, which runs it) reports "already at the latest version" and keeps the old copy. Reinstalling is the upgrade path: follow [Update the plugin](../start/quickstart-claude-code.md#update-the-plugin).
76
+
77
+ ## Start and verify
78
+
79
+ 1. Start the hub (or its service), then the Runtime with `kxm runtime start`.
80
+ 2. Check `kxm hub view`, `/ready` and `/metrics`.
81
+ 3. Check `kxm runtime status` until every project reports `ok` or `no_hub`.
82
+ 4. Send one request between two agents and read one Runtime run with `kxm runs status <run-id>`.
83
+ 5. Restart long-lived workers and reload Claude Code (`/reload-plugins`) so every agent runs the matching release.
84
+
85
+ ## Understand schema changes
86
+
87
+ Each store records a schema version, and KXM never upgrades a store in place.
88
+
89
+ | Store | Location | Created by |
90
+ |---|---|---|
91
+ | Hub store | `.kxm/state/kxm.db` (`KXM_DATA_PATH`) | `kxm hub start` |
92
+ | Runtime registry | `runtime/registry.db` under the user state root | The Runtime supervisor |
93
+ | Run event store | `runtime/projects/<key>/run-events.db` under the user state root | The Runtime supervisor |
94
+
95
+ - A store **newer** than the running release refuses to open (`runtime_schema_newer`). Do not delete it; run the release that created it, or upgrade.
96
+ - A store **older** than the running release also refuses to open (`runtime_schema_outdated`), and the error names the store. There is no migration.
97
+
98
+ When a release changes a schema, either stay on the old release, or accept a fresh store: stop the owner, move the old file aside (keep it with your backup), and start the owner, which re-creates it. `kxm init` rebuilds no database. A fresh hub store loses message and workflow history; a fresh Runtime store loses local run history. Check the changelog for schema changes before you upgrade.
99
+
100
+ > [!CAUTION]
101
+ > Deleting a store is irreversible for the history it holds. Move it aside instead, and keep the pre-upgrade backup until the new release has run cleanly.
102
+
103
+ ## Roll back
104
+
105
+ 1. Stop the Runtime, the hub and the workers.
106
+ 2. Reinstall the earlier release the same way you upgraded (for npm, `npm install --global --omit=peer @kontextmind/kxm@<version>`).
107
+ 3. Restore the pre-upgrade backup of every store the new release touched. Never open a newer-schema store with an older release.
108
+ 4. Start and verify as above.
109
+
110
+ ## Troubleshooting
111
+
112
+ | Symptom | Cause | Fix |
113
+ |---|---|---|
114
+ | `install_kind_source` or `install_kind_<kind>` from `kxm update --kxm` | KXM was not installed with a global npm install | Use the command from the table above |
115
+ | `release_digest_missing` or `release_digest_mismatch` | The GitHub release has no or a different digest | Wait for a published release, or install from npm |
116
+ | `kxm_update_config_invalid` | `update.yaml` has an unknown field or a wrong type | Fix it; only `schema`, `auto` and `source` are allowed |
117
+ | The hub refuses to start with `runtime_schema_outdated`, or a Runtime project reports that its store is not readable | The store predates this release | Roll back, or move the store aside and start fresh |
118
+ | Claude still runs the old plugin | Releases never change the plugin's version, so an update keeps the cached copy | Reinstall as [Update the plugin](../start/quickstart-claude-code.md#update-the-plugin) describes |
119
+
120
+ ## Next steps
121
+
122
+ - Confirm the service is healthy: [Monitor KXM](monitoring.md)
123
+ - Recover from a failed upgrade: [Back up and restore KXM](backup-and-restore.md)
124
+ - Every `kxm update` flag: [CLI reference](../reference/cli-reference.md#kxm-update)
@@ -2,32 +2,32 @@
2
2
  schema: "kxm.doc.v1"
3
3
  id: "PROMPT-BROWSER-006"
4
4
  type: "prompt"
5
- title: "Capturing UI Section Annotations and Sending Changes to Agent"
5
+ title: "Capture UI section annotations and send changes to an agent"
6
6
  project: "kxm"
7
7
  status: "accepted"
8
8
  owner: "@operator"
9
9
  created: "2026-09-14"
10
- updated: "2026-09-15"
10
+ updated: "2026-09-23"
11
11
  authority: "instruction"
12
12
  confidence: "verified"
13
13
  summary: "Turn annotated Steel UI-section feedback into source changes and recapture verified proof."
14
14
  tags: ["browser", "annotation", "feedback", "prompt"]
15
- related: ["docs/browser-automation.md", "docs/kb/how-to-capture-and-annotate-section.md"]
15
+ related: ["docs/guides/browser-automation.md", "docs/kb/how-to-capture-and-annotate-section.md"]
16
16
  ---
17
17
 
18
- # Task Template: Capturing UI Section Annotations and Sending Changes to Agent
18
+ # Task template: capture UI section annotations and send changes to an agent
19
19
 
20
20
  ## Purpose
21
21
 
22
22
  Use this prompt when a human operator or design critic has reviewed a UI section in a Steel browser session and wants to send annotated visual change requests back to the agent for remediation.
23
23
 
24
- ## Canonical Skill References
24
+ ## Canonical skill references
25
25
 
26
26
  - `kxm-browser-annotate`
27
27
  - `kxm-browser-verify`
28
28
  - `kxm-browser-session`
29
29
 
30
- ## Parameters & Placeholders
30
+ ## Parameters and placeholders
31
31
 
32
32
  - **PROJECT_ID**: `{{PROJECT_ID}}`
33
33
  - **TASK_ID**: `{{TASK_ID}}`
@@ -40,7 +40,7 @@ Use this prompt when a human operator or design critic has reviewed a UI section
40
40
 
41
41
  ---
42
42
 
43
- ## Instructions for Agent
43
+ ## Instructions for the agent
44
44
 
45
45
  1. **Review Visual Feedback**:
46
46
  - Inspect the section screenshot at `{{SCREENSHOT_ARTIFACT}}`.