@kontextmind/kxm 0.7.0 → 0.7.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (94) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.kxm/roles/writer.yaml +2 -0
  3. package/docs/README.md +3 -0
  4. package/docs/adr/ADR-0002-browser-automation-steel-doks.md +103 -0
  5. package/docs/agent-skills.md +19 -2
  6. package/docs/browser-automation.md +116 -0
  7. package/docs/configuration.md +10 -1
  8. package/docs/getting-started.md +21 -0
  9. package/docs/kb/how-credentials-retrieved-safely.md +31 -0
  10. package/docs/kb/how-to-capture-and-annotate-section.md +60 -0
  11. package/docs/kb/how-to-connect-playwright-to-steel.md +54 -0
  12. package/docs/kb/how-to-recover-expired-session-or-orphan.md +54 -0
  13. package/docs/kb/how-to-resume-after-mfa.md +28 -0
  14. package/docs/kb/how-to-take-over-session.md +32 -0
  15. package/docs/kb/why-authentication-disappeared.md +32 -0
  16. package/docs/kb/why-automation-opened-different-browser.md +32 -0
  17. package/docs/kb/why-session-viewer-cannot-control.md +31 -0
  18. package/docs/operations.md +24 -0
  19. package/docs/prompts/browser-annotate-feedback.md +41 -0
  20. package/docs/prompts/browser-diagnose-recover.md +38 -0
  21. package/docs/prompts/browser-explore.md +42 -0
  22. package/docs/prompts/browser-repro-fix.md +48 -0
  23. package/docs/prompts/browser-start.md +41 -0
  24. package/docs/prompts/browser-takeover.md +50 -0
  25. package/docs/skills/repo-work-delivery.md +5 -0
  26. package/docs/skills.md +2 -0
  27. package/docs/troubleshooting.md +22 -1
  28. package/package.json +1 -1
  29. package/plugins/kxm/.claude-plugin/plugin.json +1 -1
  30. package/plugins/kxm/dist/cli.js +41203 -38364
  31. package/plugins/kxm/dist/core.js +57 -0
  32. package/plugins/kxm/dist/extension.js +40 -3
  33. package/plugins/kxm/dist/mcp-server.js +1 -1
  34. package/plugins/kxm/dist/runtime.js +1582 -81
  35. package/plugins/kxm/dist/server.js +129 -4
  36. package/plugins/kxm/dist/vnext-runtime-supervisor.js +221 -41
  37. package/plugins/kxm/package.json +1 -1
  38. package/plugins/kxm/skills/hints.json +30 -0
  39. package/plugins/kxm/skills/kxm-browser-annotate/SKILL.md +90 -0
  40. package/plugins/kxm/skills/kxm-browser-auth/SKILL.md +47 -0
  41. package/plugins/kxm/skills/kxm-browser-diagnostics/SKILL.md +48 -0
  42. package/plugins/kxm/skills/kxm-browser-explore/SKILL.md +48 -0
  43. package/plugins/kxm/skills/kxm-browser-session/SKILL.md +94 -0
  44. package/plugins/kxm/skills/kxm-browser-takeover/SKILL.md +87 -0
  45. package/plugins/kxm/skills/kxm-browser-verify/SKILL.md +71 -0
  46. package/plugins/kxm/skills/kxm-hub-ops/SKILL.md +9 -0
  47. package/plugins/kxm/skills/kxm-project-setup/SKILL.md +9 -2
  48. package/plugins/kxm/src/autocomplete.ts +1 -1
  49. package/plugins/kxm/src/browser.ts +603 -0
  50. package/plugins/kxm/src/cli/context-skills.ts +373 -0
  51. package/plugins/kxm/src/cli/hub.ts +614 -0
  52. package/plugins/kxm/src/cli/roles.ts +615 -0
  53. package/plugins/kxm/src/cli/system.ts +906 -0
  54. package/plugins/kxm/src/cli/tasks.ts +364 -0
  55. package/plugins/kxm/src/cli/types.ts +270 -0
  56. package/plugins/kxm/src/cli/vnext.ts +698 -0
  57. package/plugins/kxm/src/cli/workflows.ts +699 -0
  58. package/plugins/kxm/src/cli.ts +238 -3791
  59. package/plugins/kxm/src/completion-install.ts +223 -0
  60. package/plugins/kxm/src/database.ts +1 -1
  61. package/plugins/kxm/src/external-effects.ts +1 -1
  62. package/plugins/kxm/src/hub-env.ts +193 -0
  63. package/plugins/kxm/src/init-guide-setup.ts +547 -0
  64. package/plugins/kxm/src/local-snapshot.ts +1 -1
  65. package/plugins/kxm/src/mcp-server.ts +1 -1
  66. package/plugins/kxm/src/model-inventory.ts +8 -8
  67. package/plugins/kxm/src/modes.ts +348 -0
  68. package/plugins/kxm/src/protocol.ts +111 -0
  69. package/plugins/kxm/src/role.ts +335 -0
  70. package/plugins/kxm/src/runtime.ts +4 -0
  71. package/plugins/kxm/src/safety-integrity.ts +76 -0
  72. package/plugins/kxm/src/sqlite.ts +76 -0
  73. package/plugins/kxm/src/ssh-remote.ts +560 -0
  74. package/plugins/kxm/src/store.ts +1 -1
  75. package/plugins/kxm/src/subagent-control.ts +312 -0
  76. package/plugins/kxm/src/vnext-bindings.ts +1 -1
  77. package/plugins/kxm/src/vnext-config.ts +38 -1
  78. package/plugins/kxm/src/vnext-engine-command.ts +2 -0
  79. package/plugins/kxm/src/vnext-engine.ts +16 -0
  80. package/plugins/kxm/src/vnext-harness.ts +92 -22
  81. package/plugins/kxm/src/vnext-oneshot-evidence.ts +39 -7
  82. package/plugins/kxm/src/vnext-oneshot-process.ts +46 -8
  83. package/plugins/kxm/src/vnext-oneshot-producer.ts +22 -1
  84. package/plugins/kxm/src/vnext-pi-producer.ts +11 -7
  85. package/plugins/kxm/src/vnext-runtime-store.ts +1 -1
  86. package/plugins/kxm/src/vnext-runtime-supervisor.ts +5 -3
  87. package/plugins/kxm/src/workflow-tui.ts +1 -1
  88. package/plugins/kxm/src/workflow.ts +144 -0
  89. package/schemas/vnext/modes.schema.json +56 -0
  90. package/scripts/kxm-bump-version.mjs +146 -0
  91. package/scripts/kxm-hub.mjs +145 -3
  92. package/scripts/kxm-publish-npm.mjs +3 -1
  93. package/scripts/kxm-release-github.mjs +3 -1
  94. package/scripts/kxm.mjs +0 -0
@@ -11,7 +11,7 @@
11
11
  "name": "kxm",
12
12
  "source": "./plugins/kxm",
13
13
  "description": "Durable workflows, peer agents, and kxm tui",
14
- "version": "0.7.0",
14
+ "version": "0.7.10",
15
15
  "category": "development",
16
16
  "tags": ["kxm", "multi-agent", "workflows", "mcp"]
17
17
  }
@@ -1,5 +1,7 @@
1
1
  schema: kxm.role.v1
2
2
  id: writer
3
3
  roster:
4
+ - model: xai/grok-4.6
5
+ enabled: true
4
6
  - model: openrouter/qwen/qwen3-coder-plus
5
7
  enabled: true
package/docs/README.md CHANGED
@@ -9,6 +9,8 @@ This documentation is organized by task. Start with the guide that matches what
9
9
  | [Configuration](configuration.md) | Users and operators | Understand every supported setting and default |
10
10
  | [Architecture](architecture.md) | Maintainers and integrators | Learn the component boundaries and message lifecycle |
11
11
  | [Agent Skills](agent-skills.md) | Users and integrators | Comprehensive skill suite covering all KXM commands with progressive disclosure |
12
+ | [Browser automation](browser-automation.md) | Developers and operators | Self-hosted Steel on DOKS, agent-browser, Playwright, pass-cli, and human takeover |
13
+ | [Skills](skills.md) | Operators and skill authors | Governed candidate lifecycle; also the [repository work delivery](skills/repo-work-delivery.md) skill |
12
14
  | [Operations](operations.md) | Hub operators | Run, monitor, secure, and recover the service |
13
15
  | [Troubleshooting](troubleshooting.md) | Everyone | Diagnose common installation and delivery failures |
14
16
  | [Test matrix](test-matrix.md) | Users and maintainers | Map features and use cases to automated evidence |
@@ -16,6 +18,7 @@ This documentation is organized by task. Start with the guide that matches what
16
18
  | [Peer provenance and quorum gates](provenance-gates.md) | Workflow authors and security reviewers | Require durable replies from eligible peer identities without overstating the trust guarantee |
17
19
  | [Continuous improvement](continuous-improvement.md) | Product and engineering leads | Turn run evidence into reviewed workflow improvements |
18
20
  | [Workflow guide](workflow-guide.md) | Workflow designers and operators | Area -> Workflow -> Stage -> Role taxonomy with documentation slugs, dated research candidates, and selection policy |
21
+ | [Templates](templates/README.md) | Workflow authors | Markdown templates for features, ADRs, reviews, runbooks, and related artifacts |
19
22
  | [Agent Envelopes & Quality Gates](agent-communication-envelopes-and-gates.md) | Multi-agent workflow engineers | Production communication envelopes, quality gates, and work loops |
20
23
  | [Assignment runner](assignment-runner.md) | Maintainers and developers | Native developer assignments, deterministic witness verification, and multi-vendor dual-critic acceptance |
21
24
  | [This host's Pi packages](operator-pi-packages.md) | Maintainers on this development host | Snapshot of operator `pi list` packages and file extensions; not a KXM install requirement |
@@ -0,0 +1,103 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "ADR-0002"
4
+ type: "adr"
5
+ title: "Self-Hosted Steel on DOKS for Reusable Browser Automation and Human Takeover"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "decision"
12
+ confidence: "verified"
13
+ summary: "Adopt self-hosted Steel on DigitalOcean Kubernetes (DOKS) with agent-browser and Playwright as KXM's primary browser automation infrastructure."
14
+ tags: ["architecture", "decision", "browser", "steel", "doks", "playwright"]
15
+ related: ["docs/browser-automation.md", "docs/agent-skills.md"]
16
+ details:
17
+ decision_drivers:
18
+ - "Eliminate per-minute SaaS browser provider costs"
19
+ - "Support unified same-session human takeover for MFA and sensitive authentication"
20
+ - "Provide dual exploratory (agent-browser) and regression (Playwright) interfaces"
21
+ - "Enforce strict credential isolation via pass-cli"
22
+ supersedes: null
23
+ superseded_by: null
24
+ ---
25
+
26
+ # ADR-0002: Self-Hosted Steel on DOKS for Reusable Browser Automation
27
+
28
+ ## Context & Problem Statement
29
+
30
+ AI coding agents and orchestration workflows in KXM require browser interaction for UI exploration, DOM mapping, bug reproduction, and end-to-end regression testing. Existing approaches suffered from three core issues:
31
+
32
+ 1. **High SaaS Costs**: Commercial cloud providers (e.g. Browserbase) charge steep per-session and per-minute pricing that conflicts with KXM's low-cost operating priority.
33
+ 2. **Disconnected Human Takeover**: When login challenges, MFA prompts, or CAPTCHAs occur, local or headless cloud browsers cannot easily hand the live session over to a human operator and seamlessly resume without destroying session state.
34
+ 3. **Tool Fragmentation**: Exploratory navigation needs a fast, token-efficient terminal CLI (`agent-browser`), while testing needs durable, assertion-rich frameworks (`Playwright`).
35
+
36
+ ## Decision Drivers
37
+
38
+ 1. **Operating Cost Control**: Keep infrastructure expenses predictable by utilizing our existing DigitalOcean Kubernetes Service (DOKS) cluster (`k8s-agentic-hub`).
39
+ 2. **Unified Same-Session Takeover**: Enable a human to interact with the exact same browser tab and session state during authentication gates before handing control back to the agent.
40
+ 3. **Dual Automation Interfaces**: Support `agent-browser` for discovery and `Playwright` for permanent regression tests over standard Chrome DevTools Protocol (CDP).
41
+ 4. **Authoritative Credential Management**: Ensure `pass-cli` remains the exclusive source of truth for secrets and API keys.
42
+
43
+ ## Considered Options
44
+
45
+ - **Option A**: Self-hosted Steel (`steel-dev/steel-browser`) deployed on DOKS with Ingress-NGINX and TLS.
46
+ - **Option B**: Paid SaaS browser providers (e.g., Browserbase, Steel Cloud).
47
+ - **Option C**: Local headless Chrome instances spawned on developer workstations.
48
+
49
+ ## Evaluation & Tradeoff Matrix
50
+
51
+ ### Option A: Self-hosted Steel on DOKS (Chosen)
52
+
53
+ - **Good, because**: Zero marginal per-session fees; fully self-hosted on our Kubernetes cluster.
54
+ - **Good, because**: Built-in REST API, CDP WebSocket proxy, and live session viewer UI (`/ui`).
55
+ - **Good, because**: Both `Playwright` and `agent-browser` connect seamlessly over standard CDP (`wss://steel.kontextmind.com/v1/devtools`).
56
+ - **Good, because**: Dedicated shared memory (`/dev/shm`) and resource limits prevent workstation degradation.
57
+ - **Bad, because**: Requires managing Kubernetes deployment and periodic orphaned session sweeping.
58
+
59
+ ### Option B: Paid SaaS Browser Provider
60
+
61
+ - **Good, because**: Managed scaling and proxy pools.
62
+ - **Bad, because**: Violates the core cost-efficiency constraint; introduces recurring credit card charges and third-party data transmission risks.
63
+
64
+ ### Option C: Local Chrome Instances
65
+
66
+ - **Good, because**: No cluster deployment needed.
67
+ - **Bad, because**: High workstation memory and CPU pressure; fragile cross-platform headless setups; cannot easily share live debug sessions across multi-agent environments.
68
+
69
+ ## Decision Outcome
70
+
71
+ **Chosen Option**: **Option A (Self-hosted Steel on DOKS)**.
72
+
73
+ ### Architecture
74
+
75
+ ```text
76
+ ┌─────────────────────────────────────────────────────────────┐
77
+ │ KXM Agent / Herdr │
78
+ │ (kxm-browser-session, kxm-browser-takeover, pass-cli) │
79
+ └───────────────┬─────────────────────────────┬───────────────┘
80
+ │ REST API (create/release) │ CDP WebSocket
81
+ ▼ ▼
82
+ ┌─────────────────────────────────────────────────────────────┐
83
+ │ DigitalOcean Kubernetes (DOKS) │
84
+ │ https://steel.kontextmind.com │
85
+ │ │
86
+ │ ┌─────────────────────┐ ┌────────────────────────┐ │
87
+ │ │ Steel API & CDP │◄─────►│ Chromium Sandbox │ │
88
+ │ │ (Fastify) │ │ (/dev/shm 2Gi) │ │
89
+ │ └──────────┬──────────┘ └────────────────────────┘ │
90
+ │ │ │
91
+ │ ▼ │
92
+ │ ┌─────────────────────┐ │
93
+ │ │ Session Viewer │ ◄─── Human Takeover (MFA/Auth) │
94
+ │ │ (/ui) │ │
95
+ │ └─────────────────────┘ │
96
+ └─────────────────────────────────────────────────────────────┘
97
+ ```
98
+
99
+ ## Confirmation & Verification Strategy
100
+
101
+ - **Verification**: Health endpoint `https://steel.kontextmind.com/v1/health` verified with HTTP 200 and Let's Encrypt TLS.
102
+ - **Integration Test**: `test/core/browser.test.ts` validates session lifecycle, CDP endpoint formatting, takeover transitions, and secret redaction.
103
+ - **Security Check**: `pass-cli` verified as the authoritative store for `STEEL_API_KEY` under vault `AI Provider Keys`.
@@ -10,7 +10,7 @@ authority, admit new writers, or replace trusted `.kxm/roster.json` policy.
10
10
  | Feature Area | Skill | Commands Covered | Purpose |
11
11
  |---|---|---|---|
12
12
  | Core Routing | `kxm` | — | Select the right suite skill; state universal safety rules and portable CLI convention |
13
- | Project Setup | `kxm-project-setup` | `init`, `migrate`, `trust`, `config`, `completion` | Initialize, migrate, review permission changes, configure, and add shell completion |
13
+ | Project Setup | `kxm-project-setup` | `init`, `migrate`, `trust`, `config`, `completion` | Initialize, migrate, review permission changes, configure, and install shell completion |
14
14
  | Harness & Auth | `kxm-harness-auth` | `harness`, `auth`, `update`, `runtime`, `agent` | Inspect authenticated harness capability and operate supported runtimes/workers |
15
15
  | Hub Operations | `kxm-hub-ops` | `hub`, `backup`, `restore` | Run and protect the local hub and its durable SQLite state |
16
16
  | Session Management | `kxm-session` | `session`, `dash`, `studio` | Resume/inspect operator work and use UI capabilities each harness supports |
@@ -23,6 +23,20 @@ authority, admit new writers, or replace trusted `.kxm/roster.json` policy.
23
23
  | Routing & Improve | `kxm-routing-improve` | `routing`, `improve` | Inspect real route quality/cost and propose reviewed improvements |
24
24
  | Tasks & Goals | `kxm-tasks` | `suggest`, `goal`, `task` | Recommend workflows and manage goals/tasks with SCM/tracker boundaries |
25
25
 
26
+ ## Browser Automation Skills
27
+
28
+ KXM includes dedicated skills for remote browser automation on self-hosted Steel (DOKS), exploratory navigation via `agent-browser`, testing with `Playwright`, and visual feedback. See [Browser Automation Guide](browser-automation.md) and [ADR-0002](adr/ADR-0002-browser-automation-steel-doks.md).
29
+
30
+ | Feature Area | Skill | Purpose |
31
+ |---|---|---|
32
+ | Browser Sessions | `kxm-browser-session` | Start, attach, inspect, and release Steel sessions on DOKS |
33
+ | Human Takeover | `kxm-browser-takeover` | Handoff protocol for MFA, login, CAPTCHA, and sensitive consent |
34
+ | Credentials & Profiles | `kxm-browser-auth` | Retrieve credentials from `pass-cli` and manage authenticated profiles safely |
35
+ | Exploration | `kxm-browser-explore` | Exploratory navigation, DOM inspection, and workflow mapping via `agent-browser` |
36
+ | Reproduction & Verify | `kxm-browser-verify` | Reproduce UI bugs, gather evidence, and author durable Playwright tests |
37
+ | Diagnostics & Recovery | `kxm-browser-diagnostics` | Investigate Steel connectivity, CDP errors, timeouts, and orphan cleanup |
38
+ | Section Annotation | `kxm-browser-annotate` | Capture DOM sections, attach structured annotations, and feed changes to agents |
39
+
26
40
  ## Installation and Discovery
27
41
 
28
42
  ### For Pi Users
@@ -89,7 +103,10 @@ The skills in this suite (`kxm-*`) are bundled and operational by default. They
89
103
 
90
104
  Separately, `kxm skills` manages community or experimental candidates through
91
105
  create/evaluate/promote/reject/verify. Those governed skills are distinct from
92
- this bundled suite. Telemetry cannot auto-promote a skill.
106
+ this bundled suite. Telemetry cannot auto-promote a skill. See
107
+ [Skill candidate lifecycle](skills.md) for the full lifecycle, and
108
+ [Repository work delivery](skills/repo-work-delivery.md) for converting a
109
+ repository request into a delivery prompt.
93
110
 
94
111
  ## Development and Maintenance
95
112
 
@@ -0,0 +1,116 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "GUIDE-BROWSER-001"
4
+ type: "guide"
5
+ title: "KXM Browser Automation Guide"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Comprehensive guide to browser automation in KXM using self-hosted Steel on DOKS, agent-browser, Playwright, pass-cli, and human takeover."
14
+ tags: ["browser", "automation", "steel", "playwright", "agent-browser", "doks"]
15
+ related: ["docs/adr/ADR-0002-browser-automation-steel-doks.md", "docs/agent-skills.md"]
16
+ ---
17
+
18
+ # KXM Browser Automation Guide
19
+
20
+ This guide describes how to use KXM's browser automation capability powered by self-hosted Steel on DigitalOcean Kubernetes (DOKS), `agent-browser` for exploratory inspection, `Playwright` for automated regression testing, and `pass-cli` for credential security.
21
+
22
+ ---
23
+
24
+ ## 1. Architecture & Trust Boundaries
25
+
26
+ ```text
27
+ Herdr / Pi / Agent Harness
28
+ │
29
+ ├──► KXM Browser Skills & Prompts (kxm-browser-*)
30
+ │
31
+ ├──► pass-cli (Authoritative Vault: "AI Provider Keys")
32
+ │
33
+ ├──► agent-browser (Exploratory CLI) ──┐
34
+ │ │ (CDP WebSocket)
35
+ ├──► Playwright (E2E Test Suites) ────┼──► DOKS Steel Browser Cluster
36
+ │ │ (https://steel.kontextmind.com)
37
+ └──► Human Operator (Takeover UI) ─────┘ (https://steel.kontextmind.com/ui)
38
+ ```
39
+
40
+ ### Key Components
41
+
42
+ 1. **Steel on DOKS (`https://steel.kontextmind.com`)**:
43
+ - Primary browser execution environment.
44
+ - Isolated Chromium containers with dedicated shared memory (`/dev/shm`).
45
+ - Exposed endpoints: REST API (port 443 / 3000), CDP WebSocket proxy (port 443 / 9223), Web UI (`/ui`), OpenAPI docs (`/documentation`).
46
+ 2. **`pass-cli`**:
47
+ - The authoritative store for all long-lived passwords, session tokens, and the `STEEL_API_KEY`.
48
+ - Never write credentials to tracked git files.
49
+ 3. **`agent-browser`**:
50
+ - Fast, token-efficient terminal CLI for accessibility snapshots, interactive navigation, and exploratory testing.
51
+ 4. **`Playwright`**:
52
+ - High-fidelity assertion engine for bug reproduction, visual proofs, and permanent regression suites.
53
+
54
+ ---
55
+
56
+ ## 2. Getting Started: The First Browser Workflow
57
+
58
+ ### Step 1: Verify Prerequisites
59
+
60
+ Check that `pass-cli` can access the Steel configuration:
61
+
62
+ ```bash
63
+ pass-cli item view --vault-name "AI Provider Keys" --item-title "Steel Browser (KontextMind DOKS)"
64
+ ```
65
+
66
+ ### Step 2: Launch a Steel Browser Session
67
+
68
+ ```bash
69
+ STEEL_KEY=$(pass-cli item view --vault-name "AI Provider Keys" --item-title "Steel Browser (KontextMind DOKS)" --field STEEL_API_KEY)
70
+
71
+ # Create a 5-minute session
72
+ SESSION_RESP=$(curl -s -X POST https://steel.kontextmind.com/v1/sessions \
73
+ -H "Content-Type: application/json" \
74
+ -H "x-steel-api-key: $STEEL_KEY" \
75
+ -d '{"timeout": 300000}')
76
+
77
+ SESSION_ID=$(echo "$SESSION_RESP" | jq -r .id)
78
+ echo "Session created: $SESSION_ID"
79
+ ```
80
+
81
+ ### Step 3: Run Exploratory Automation or Scrapes
82
+
83
+ To run a fast scrape without managing sessions:
84
+
85
+ ```bash
86
+ curl -s -X POST https://steel.kontextmind.com/v1/scrape \
87
+ -H "Content-Type: application/json" \
88
+ -H "x-steel-api-key: $STEEL_KEY" \
89
+ -d '{"url": "https://example.com"}'
90
+ ```
91
+
92
+ ### Step 4: Release the Session
93
+
94
+ ```bash
95
+ curl -s -X POST https://steel.kontextmind.com/v1/sessions/$SESSION_ID/release \
96
+ -H "x-steel-api-key: $STEEL_KEY"
97
+ ```
98
+
99
+ ---
100
+
101
+ ## 3. Human Takeover Protocol
102
+
103
+ When authentication challenges (MFA, CAPTCHA, SSO) are encountered:
104
+
105
+ 1. **Agent Pauses**: The agent stops automated actions and sets state to `HUMAN_CONTROL`.
106
+ 2. **Emits Link**: Emits the protected URL: `https://steel.kontextmind.com/ui?sessionId=<SESSION_ID>`.
107
+ 3. **Human Interacts**: The operator opens the session viewer, completes the login/MFA action, and confirms in the terminal (`auth complete`).
108
+ 4. **Agent Verifies**: The agent inspects the resulting page, confirms dashboard/user state, refreshes DOM observations, and returns to `AGENT_CONTROL`.
109
+
110
+ ---
111
+
112
+ ## 4. Resource Controls & Cost Safety
113
+
114
+ - **Session Timeouts**: Default 300s (5m), max 1800s (30m). Sessions terminate automatically on expiry.
115
+ - **Orphan Sweeping**: Periodically sweep untracked sessions via `GET /v1/sessions` and release idle processes.
116
+ - **Memory Protection**: Kubernetes mounts a 2Gi `emptyDir` memory volume at `/dev/shm` to prevent Chromium tab crashes without exhausting node memory.
@@ -74,6 +74,15 @@ rewrites it.
74
74
 
75
75
  The hub refuses a non-loopback bind without `KXM_AUTH_TOKEN`. Use a long random administrative token even when project tokens are configured, because administrative endpoints such as `/metrics` require it outside loopback.
76
76
 
77
+ When `KXM_AUTH_TOKEN` is unset, `kxm hub start` resolves credentials from the
78
+ persisted `kxm.hub-env.v1` file under the user state root
79
+ (`~/.local/state/kxm/hub-env.json`; honors `KXM_STATE_HOME` and platform
80
+ equivalents). A missing admin token is generated once, persisted with `0600`
81
+ permissions, and reused across hub restarts so workers and dashboards on the
82
+ same machine share one stable credential. Explicit `KXM_AUTH_TOKEN` or
83
+ `KXM_PROJECT_TOKENS` environment values take precedence and are persisted for
84
+ later restarts. Delete the file and restart to rotate the generated token.
85
+
77
86
  Project tokens are an authorization boundary. A project-specific token can register only in its mapped project and see only that project's agents and messages. The administrative token remains a fallback for projects without an explicit entry. For provenance-gated workflows, configure an explicit project token and give workers only that token; reserve a distinct administrative token for operations such as quorum degradation approval.
78
87
 
79
88
  PowerShell example:
@@ -279,7 +288,7 @@ See [Webhook workflows](webhook-workflows.md) for the base schema and the comple
279
288
  | `steer` | An active blocker requires a course change | Deliver at the next decision boundary |
280
289
  | `nextTurn` | Information should wait for a later turn | Queue context without immediate work |
281
290
 
282
- `followUp` is the safe default. Use [`.kxm/env.example`](../ .kxm/env.example) as a reference, but load values through your shell, supervisor, container platform, or secret manager. Never commit real tokens.
291
+ `followUp` is the safe default. Load values through your shell, supervisor, container platform, or secret manager using the variables documented on this page and in the [KXM Handbook](kxm-handbook.md). Never commit real tokens.
283
292
 
284
293
  ## Nous providers (opt-in)
285
294
 
@@ -61,6 +61,27 @@ kxm init
61
61
 
62
62
  `kxm init` never copies the package repository's dogfood roster or workflows into a consumer workspace.
63
63
 
64
+ When `kxm init` succeeds in an interactive terminal, it offers to install shell
65
+ completion for the detected shell. Accepting writes the completion script
66
+ under the user config directory, appends one idempotent stanza to the shell
67
+ rc file, and, when the kxm bin directory is not already on `PATH`, adds a
68
+ `PATH` export. Declining is safe: run `kxm completion install` later, or set
69
+ `KXM_SKIP_COMPLETION_PROMPT=1` to suppress the offer. Non-interactive,
70
+ `--json`, and `--dry-run` runs never prompt or write shell files.
71
+
72
+ After the completion offer, an interactive `kxm init` also offers to set up
73
+ workflow-guide agents and workflows for the harnesses you have installed and
74
+ authenticated. Accepting lists the software-engineering workflows from
75
+ [`workflow-guide.md`](workflow-guide.md); pick by number or slug (`all` works
76
+ too). kxm resolves each role's first guide candidate whose harness is
77
+ authenticated and writes only current vNext project resources —
78
+ `.kxm/agents/<role>.yaml` (`kxm.agent.v1`) and `.kxm/workflows/<slug>.yaml`
79
+ (`kxm.workflow.v1`). It never writes retired legacy authority (`.kxm/config`,
80
+ `.kxm/roster.json`). Roles whose candidates have no authenticated harness are
81
+ reported as skipped, not silently downgraded. Guide candidates are dated
82
+ research — verify them before dispatch. Declining is safe: set
83
+ `KXM_SKIP_GUIDE_SETUP_PROMPT=1` to suppress the offer.
84
+
64
85
  ## 3. Start the hub in another terminal
65
86
 
66
87
  `kxm hub start` is foreground. Keep that terminal running.
@@ -0,0 +1,31 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-003"
4
+ type: "kb"
5
+ title: "How are credentials retrieved without exposing them to the model?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Explains safe pass-cli credential delivery and environment piping patterns that avoid LLM context leakage."
14
+ tags: ["browser", "credentials", "pass-cli", "security"]
15
+ related: ["docs/browser-automation.md", "docs/agent-skills.md"]
16
+ ---
17
+
18
+ # How are credentials retrieved without exposing them to the model?
19
+
20
+ To protect passwords, MFA secrets, and API tokens from leaking into model reasoning traces, KXM enforces strict credential-reference boundaries:
21
+
22
+ ## Mechanisms
23
+
24
+ 1. **Authoritative Store**: All secrets reside in `pass-cli` (Proton Pass).
25
+ 2. **In-Process Environment Piping**:
26
+ - Automated test scripts use `pass-cli run -- npm test` or retrieve credentials directly into child process memory via standard environment variables.
27
+ - The LLM prompt only receives credential references (e.g., `vault: "AI Provider Keys", item: "Steel Browser (KontextMind DOKS)"`), never raw secret values.
28
+ 3. **Log Sanitization**:
29
+ - The KXM browser client strips API keys and token parameters (`apiKey=[REDACTED]`, `steel_[REDACTED]`) before logging or emitting outputs.
30
+ 4. **Human Handoff for High-Privilege Auth**:
31
+ - For sensitive production accounts or MFA, the agent never touches the credential at all; it invokes `kxm-browser-takeover` and lets the human authenticate directly in the UI.
@@ -0,0 +1,60 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-009"
4
+ type: "kb"
5
+ title: "How do I capture a UI section and annotate changes for an agent?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Guide to capturing specific DOM elements, attaching visual annotations, and submitting structured change feedback to the agent."
14
+ tags: ["browser", "annotation", "screenshot", "feedback", "ui"]
15
+ related: ["docs/browser-automation.md", "docs/prompts/browser-annotate-feedback.md"]
16
+ ---
17
+
18
+ # How do I capture a UI section and annotate changes for an agent?
19
+
20
+ When reviewing a web interface in a Steel session, you can isolate a specific component, attach annotations, and deliver structured change requests directly back to an agent.
21
+
22
+ ## Workflow
23
+
24
+ 1. **Capture the Component**:
25
+ Use Playwright element screenshotting to crop only the affected container:
26
+
27
+ ```typescript
28
+ await page.locator('.billing-card').screenshot({ path: '.kxm/artifacts/browser/billing-card.png' });
29
+ ```
30
+
31
+ 2. **Draft the Annotation Feedback**:
32
+ Record the target selector, observed issues, and required fixes:
33
+
34
+ ```typescript
35
+ import { createAnnotationFeedback, formatAnnotationFeedbackPrompt } from "@kontextmind/kxm/runtime";
36
+
37
+ const feedback = createAnnotationFeedback({
38
+ url: "https://app.example.com/settings/billing",
39
+ sectionSelector: ".billing-card",
40
+ screenshotPath: ".kxm/artifacts/browser/billing-card.png",
41
+ overallSummary: "Billing tier layout breaks on mobile viewport",
42
+ annotations: [
43
+ {
44
+ label: "Tier Name Overflow",
45
+ selector: ".tier-title",
46
+ note: "Truncate or wrap long tier titles with ellipsis",
47
+ severity: "fix"
48
+ }
49
+ ],
50
+ requestedChanges: [
51
+ "Update `.tier-title` CSS to include `truncate` or `break-words`",
52
+ "Adjust padding on small viewports to `px-4`"
53
+ ]
54
+ });
55
+
56
+ const prompt = formatAnnotationFeedbackPrompt(feedback);
57
+ ```
58
+
59
+ 3. **Send to the Agent**:
60
+ Feed the rendered prompt into the agent session or KXM workflow run. The agent reads the screenshot, navigates to the source code, applies the changes, and verifies the result with Playwright.
@@ -0,0 +1,54 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-007"
4
+ type: "kb"
5
+ title: "How do I connect Playwright to the existing Steel session?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Guide to attaching Playwright tests to an active remote Steel browser session using chromium.connectOverCDP()."
14
+ tags: ["browser", "playwright", "cdp", "steel", "testing"]
15
+ related: ["docs/browser-automation.md", "docs/kb/why-automation-opened-different-browser.md"]
16
+ ---
17
+
18
+ # How do I connect Playwright to the existing Steel session?
19
+
20
+ To run Playwright tests against self-hosted Steel on DOKS instead of a local browser:
21
+
22
+ ## 1. Retrieve the CDP Endpoint
23
+
24
+ Format the WebSocket CDP URL using the active session ID and API key:
25
+
26
+ ```typescript
27
+ import { formatCDPEndpoint, resolveSteelConfig } from "@kontextmind/kxm/runtime";
28
+
29
+ const config = resolveSteelConfig();
30
+ const cdpUrl = formatCDPEndpoint({ id: sessionId, websocketUrl: "" }, config);
31
+ ```
32
+
33
+ The resulting URL will look like:
34
+ `wss://steel.kontextmind.com/v1/devtools?sessionId=<SESSION_ID>&apiKey=<STEEL_API_KEY>`
35
+
36
+ ## 2. Connect in Playwright
37
+
38
+ ```typescript
39
+ import { test, expect, chromium } from "@playwright/test";
40
+
41
+ test("execute test on remote steel session", async () => {
42
+ const browser = await chromium.connectOverCDP(process.env.STEEL_CDP_URL!);
43
+
44
+ // Use existing context or create one
45
+ const context = browser.contexts()[0] || await browser.newContext();
46
+ const page = context.pages()[0] || await context.newPage();
47
+
48
+ await page.goto("https://app.example.com");
49
+ await expect(page.getByRole("heading", { level: 1 })).toBeVisible();
50
+
51
+ // Disconnecting closes the Playwright CDP socket without terminating the remote container
52
+ await browser.close();
53
+ });
54
+ ```
@@ -0,0 +1,54 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-008"
4
+ type: "kb"
5
+ title: "How do I recover an expired session or remove an orphaned browser?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Procedures for detecting and releasing stale or orphaned browser sessions on Steel."
14
+ tags: ["browser", "cleanup", "orphans", "troubleshooting"]
15
+ related: ["docs/browser-automation.md", "docs/kb/why-authentication-disappeared.md"]
16
+ ---
17
+
18
+ # How do I recover an expired session or remove an orphaned browser?
19
+
20
+ If an automation run crashed or disconnected without calling `/release`, a browser container may remain idling on DOKS.
21
+
22
+ ## 1. List Active Remote Sessions
23
+
24
+ ```bash
25
+ STEEL_KEY=$(pass-cli item view --vault-name "AI Provider Keys" --item-title "Steel Browser (KontextMind DOKS)" --field STEEL_API_KEY)
26
+
27
+ curl -s https://steel.kontextmind.com/v1/sessions \
28
+ -H "x-steel-api-key: $STEEL_KEY" | jq .
29
+ ```
30
+
31
+ ## 2. Release Orphaned Sessions
32
+
33
+ To terminate a specific stale session:
34
+
35
+ ```bash
36
+ curl -s -X POST https://steel.kontextmind.com/v1/sessions/<SESSION_ID>/release \
37
+ -H "x-steel-api-key: $STEEL_KEY"
38
+ ```
39
+
40
+ ## 3. Automatic Orphan Sweeping via KXM Client
41
+
42
+ The KXM client provides `checkOrphanedSessions(maxIdleMs)` to automate this:
43
+
44
+ ```typescript
45
+ import { SteelClient } from "@kontextmind/kxm/runtime";
46
+
47
+ const client = new SteelClient();
48
+ const orphans = await client.checkOrphanedSessions(600000); // > 10 min idle
49
+
50
+ for (const sessionId of orphans) {
51
+ console.log(`Releasing orphaned session: ${sessionId}`);
52
+ await client.releaseSession(sessionId);
53
+ }
54
+ ```
@@ -0,0 +1,28 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-002"
4
+ type: "kb"
5
+ title: "How does an agent resume after MFA?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Details the verification and observation refresh sequence when resuming automation after MFA."
14
+ tags: ["browser", "mfa", "resume", "verification"]
15
+ related: ["docs/kb/how-to-take-over-session.md", "docs/browser-automation.md"]
16
+ ---
17
+
18
+ # How does an agent resume after MFA?
19
+
20
+ Once the human completes the MFA challenge in the browser viewer, the agent must not immediately execute blind clicks. It follows this sequence:
21
+
22
+ 1. **State Transition**: Moves from `HUMAN_CONTROL` to `VERIFY_AUTHENTICATION`.
23
+ 2. **CDP Re-attachment**: Re-queries the active tab target from the remote Steel CDP endpoint.
24
+ 3. **App State Verification**:
25
+ - Inspects `page.url()` to confirm the browser navigated away from the MFA prompt to the intended destination (e.g. `/dashboard` or `/overview`).
26
+ - Checks for authenticated elements (e.g. account menu, logout button, user profile avatar).
27
+ 4. **Observation Refresh**: Runs a fresh `agent-browser snapshot` or queries fresh DOM locators before executing the next action.
28
+ 5. **Restore Control**: Returns to `AGENT_CONTROL` and proceeds with the task.
@@ -0,0 +1,32 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-001"
4
+ type: "kb"
5
+ title: "How do I take over a browser session to log in?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Instructions for taking over an active Steel browser session during an authentication gate."
14
+ tags: ["browser", "takeover", "auth", "mfa"]
15
+ related: ["docs/browser-automation.md", "docs/kb/how-to-resume-after-mfa.md"]
16
+ ---
17
+
18
+ # How do I take over a browser session to log in?
19
+
20
+ When an agent encounters a login screen, OAuth prompt, or security challenge, it triggers the `kxm-browser-takeover` protocol.
21
+
22
+ ## Steps
23
+
24
+ 1. **Copy the Takeover URL**:
25
+ The agent will emit a message in the terminal with a link like:
26
+ `https://steel.kontextmind.com/ui?sessionId=<SESSION_ID>`
27
+ 2. **Open the Session Viewer**:
28
+ Open that URL in your desktop browser. You will see the live screencast of the exact remote Chrome container the agent was operating.
29
+ 3. **Interact and Authenticate**:
30
+ Enter the username, password, or security key into the session, or use the devtools inspector (`https://steel.kontextmind.com/v1/devtools/inspector.html`) to trigger the submission.
31
+ 4. **Signal Completion**:
32
+ Return to your Herdr or Pi terminal session and notify the agent: `auth complete` or `proceed`.
@@ -0,0 +1,32 @@
1
+ ---
2
+ schema: "kxm.doc.v1"
3
+ id: "KB-BROWSER-004"
4
+ type: "kb"
5
+ title: "Why did authentication disappear?"
6
+ project: "kxm"
7
+ status: "accepted"
8
+ owner: "@operator"
9
+ created: "2026-09-14"
10
+ updated: "2026-09-14"
11
+ authority: "instruction"
12
+ confidence: "verified"
13
+ summary: "Diagnosing lost authentication state, session expiration, and ephemeral container recreation."
14
+ tags: ["browser", "authentication", "cookies", "troubleshooting"]
15
+ related: ["docs/browser-automation.md", "docs/kb/why-automation-opened-different-browser.md"]
16
+ ---
17
+
18
+ # Why did authentication disappear?
19
+
20
+ If an agent was authenticated on a previous step or run and suddenly encounters a login screen again, the root causes are typically:
21
+
22
+ ## Root Causes & Fixes
23
+
24
+ 1. **Session Released or Expired**:
25
+ - Steel sessions are ephemeral by default. Once a session reaches its timeout (e.g. 5–30 minutes) or is released via `POST /v1/sessions/:id/release`, all memory cookies and local storage are cleared.
26
+ - **Fix**: To reuse state across tasks, save `storageState` via Playwright and reload it on the next session initialization.
27
+ 2. **New Session Launched Instead of Attaching**:
28
+ - If the agent created a brand-new session instead of passing the existing `sessionId`, it opened a clean Chrome profile.
29
+ - **Fix**: Verify that the task passes `sessionId` to `getSession()` or uses the existing CDP endpoint.
30
+ 3. **Domain / Subdomain Cookie Scoping**:
31
+ - OAuth flows often set cookies on subdomains (e.g., `auth.example.com`) that do not automatically share with `app.example.com`.
32
+ - **Fix**: Ensure cookies were issued for the primary domain or that SSO redirect completed fully before saving state.