@xenosystem/agent-sdk 0.9.11 → 0.9.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/LICENSE +15 -15
  2. package/README.md +511 -511
  3. package/dist/agents/index.cjs +5 -5
  4. package/dist/agents/index.js +15 -15
  5. package/dist/agents/metafile-cjs.json +1 -1
  6. package/dist/agents/metafile-esm.json +1 -1
  7. package/dist/artifacts/index.cjs +1 -1
  8. package/dist/artifacts/index.js +1 -1
  9. package/dist/artifacts/metafile-cjs.json +1 -1
  10. package/dist/artifacts/metafile-esm.json +1 -1
  11. package/dist/automation/index.d.cts +1 -0
  12. package/dist/automation/index.d.ts +1 -0
  13. package/dist/automation/metafile-cjs.json +1 -1
  14. package/dist/automation/metafile-esm.json +1 -1
  15. package/dist/control-plane/index.cjs +1 -1
  16. package/dist/control-plane/index.js +1 -1
  17. package/dist/control-plane/metafile-cjs.json +1 -1
  18. package/dist/control-plane/metafile-esm.json +1 -1
  19. package/dist/control-room/metafile-cjs.json +1 -1
  20. package/dist/control-room/metafile-esm.json +1 -1
  21. package/dist/electron/index.cjs +107 -107
  22. package/dist/electron/index.d.cts +8 -0
  23. package/dist/electron/index.d.ts +8 -0
  24. package/dist/electron/index.js +103 -103
  25. package/dist/electron/metafile-cjs.json +1 -1
  26. package/dist/electron/metafile-esm.json +1 -1
  27. package/dist/governance/index.cjs +18 -18
  28. package/dist/governance/index.d.cts +1 -0
  29. package/dist/governance/index.d.ts +1 -0
  30. package/dist/governance/index.js +18 -18
  31. package/dist/governance/metafile-cjs.json +1 -1
  32. package/dist/governance/metafile-esm.json +1 -1
  33. package/dist/hooks/metafile-cjs.json +1 -1
  34. package/dist/hooks/metafile-esm.json +1 -1
  35. package/dist/hosted/index.cjs +1 -1
  36. package/dist/hosted/index.js +1 -1
  37. package/dist/hosted/metafile-cjs.json +1 -1
  38. package/dist/hosted/metafile-esm.json +1 -1
  39. package/dist/index.cjs +347 -329
  40. package/dist/index.d.cts +113 -2
  41. package/dist/index.d.ts +113 -2
  42. package/dist/index.js +329 -311
  43. package/dist/intelligence/metafile-cjs.json +1 -1
  44. package/dist/intelligence/metafile-esm.json +1 -1
  45. package/dist/mcp/index.d.cts +1 -0
  46. package/dist/mcp/index.d.ts +1 -0
  47. package/dist/mcp/metafile-cjs.json +1 -1
  48. package/dist/mcp/metafile-esm.json +1 -1
  49. package/dist/metafile-cjs.json +1 -1
  50. package/dist/metafile-esm.json +1 -1
  51. package/dist/oracle/metafile-cjs.json +1 -1
  52. package/dist/oracle/metafile-esm.json +1 -1
  53. package/dist/providers/metafile-cjs.json +1 -1
  54. package/dist/providers/metafile-esm.json +1 -1
  55. package/dist/recipes/metafile-cjs.json +1 -1
  56. package/dist/recipes/metafile-esm.json +1 -1
  57. package/dist/research/metafile-cjs.json +1 -1
  58. package/dist/research/metafile-esm.json +1 -1
  59. package/dist/session/index.cjs +2 -2
  60. package/dist/session/index.d.cts +8 -0
  61. package/dist/session/index.d.ts +8 -0
  62. package/dist/session/index.js +2 -2
  63. package/dist/session/metafile-cjs.json +1 -1
  64. package/dist/session/metafile-esm.json +1 -1
  65. package/dist/sharing/metafile-cjs.json +1 -1
  66. package/dist/sharing/metafile-esm.json +1 -1
  67. package/dist/skills/index.d.cts +1 -0
  68. package/dist/skills/index.d.ts +1 -0
  69. package/dist/skills/metafile-cjs.json +1 -1
  70. package/dist/skills/metafile-esm.json +1 -1
  71. package/dist/soul/metafile-cjs.json +1 -1
  72. package/dist/soul/metafile-esm.json +1 -1
  73. package/dist/terminal/metafile-cjs.json +1 -1
  74. package/dist/terminal/metafile-esm.json +1 -1
  75. package/dist/ui/index.d.cts +1 -0
  76. package/dist/ui/index.d.ts +1 -0
  77. package/dist/ui/metafile-cjs.json +1 -1
  78. package/dist/ui/metafile-esm.json +1 -1
  79. package/dist/utils/index.d.cts +1 -0
  80. package/dist/utils/index.d.ts +1 -0
  81. package/dist/utils/metafile-cjs.json +1 -1
  82. package/dist/utils/metafile-esm.json +1 -1
  83. package/native/mxc/LICENSE.microsoft-mxc-sdk.md +20 -20
  84. package/native/mxc/README.md +25 -25
  85. package/native/mxc/package.json +58 -58
  86. package/ownership/CLEAN_ROOM_CONTRIBUTOR_GUIDE.md +63 -63
  87. package/ownership/evidence/PROPRIETARY-COMMAND-1/implementation-record.md +77 -77
  88. package/ownership/evidence/PROPRIETARY-COMMAND-1/provenance-review.md +51 -51
  89. package/ownership/evidence/PROPRIETARY-MEDIA-1/implementation-record.md +131 -131
  90. package/ownership/evidence/PROPRIETARY-MEDIA-1/provenance-review.md +50 -50
  91. package/ownership/evidence/PROPRIETARY-PTY-1/implementation-record.md +86 -86
  92. package/ownership/evidence/PROPRIETARY-PTY-1/provenance-review.md +52 -52
  93. package/ownership/evidence/PROPRIETARY-RUNTIME-1/implementation-record.md +68 -68
  94. package/ownership/evidence/PROPRIETARY-RUNTIME-1/provenance-review.md +32 -32
  95. package/ownership/evidence/PROPRIETARY-TOOLCHAIN-1/implementation-record.md +53 -53
  96. package/ownership/evidence/PROPRIETARY-TOOLCHAIN-1/provenance-review.md +32 -32
  97. package/ownership/evidence/PROPRIETARY-UI-1/implementation-record.md +96 -96
  98. package/ownership/evidence/PROPRIETARY-UI-1/provenance-review.md +46 -46
  99. package/ownership/ownership-policy.schema.json +192 -192
  100. package/ownership/templates/clean-room-implementation-record.md +70 -70
  101. package/ownership/templates/provenance-review.md +51 -51
  102. package/package.json +1 -1
package/README.md CHANGED
@@ -1,511 +1,511 @@
1
- # XENO Agent SDK
2
-
3
- Core runtime for XENO agent-enabled products. It provides the agent loop, tool execution, permissions, sessions, memory, audit logging, and provider integration used across the XENO platform.
4
-
5
- ## Runtime Ownership
6
-
7
- The ownership program targets P2 feature parity, but this checkout must be described by its generated audit rather than by the target. Run `npm run build && npm run compliance:ownership:artifacts` or inspect `docs/compliance/ownership-inventory.md`. Node remains an explicitly external host through P2, and development/build packages remain third-party build inputs. Do not claim that a release is fully proprietary, entirely Xeno-owned, or free of third-party runtime code unless the packaged release smoke prints the corresponding achieved level and the required provenance/counsel gates are complete.
8
-
9
- The canonical policy and engineering clean-room controls are under `ownership/`.
10
-
11
- ## Vision
12
-
13
- Every XENO creative app (Pixel, Motion, Sound) has an AI agent embedded directly into the interface. Users open a sidebar, type what they want ("remove the background from this layer", "cut the silence from this podcast", "match-cut these two clips"), and the agent translates that into tool calls against the app's engine. One request can span multiple apps: "Create a product video from these photos with background music" triggers coordinated work across Pixel, Motion, and Sound simultaneously.
14
-
15
- The SDK is designed so the same governed agent contracts can be embedded across
16
- creative products instead of rebuilding orchestration in every application.
17
-
18
- ```
19
- Without agent SDK: User manually opens Pixel, edits, opens Motion, edits, opens Sound, edits
20
- With agent SDK: User says "Create a product video" -> agents in all 3 apps coordinate automatically
21
- ```
22
-
23
- ## Current State
24
-
25
- - **Provider-agnostic runtime** with XENO-hosted, local, direct-provider, and
26
- generic compatible routes
27
- - **Extensible tool system** with demand-loaded schemas, managed background
28
- operations, terminal support, and host-registered domain tools
29
- - **4 permission modes**: `default`, `acceptEdits`, `bypassPermissions`, `plan`
30
- - **4-level identity hierarchy**: global, project, role, session
31
- - **4-level memory system** with auto-capture (conversation, session, project, global)
32
- - **Session persistence** with checkpoints, transcript writing, and crash recovery
33
- - **Audit logging** in JSON-lines format (who, what, when, result)
34
- - **Agent orchestration** with profiles, delegation, teams, goals, workflows,
35
- monitors, schedules, hooks, and durable recovery
36
- - **Artifact and evidence protocol** with immutable revisions, review events,
37
- stable anchors, provenance, and evidence graphs
38
- - **Secure execution contracts and capability leases** that preserve
39
- profile/host denials and distinguish policy, hardening, and containment
40
- - **Agent Skills v2** with interoperable `SKILL.md` discovery, progressive
41
- disclosure, hash-bound resources, policy intersection, and privacy-safe audit
42
- - **Current MCP transport and OAuth** with Streamable HTTP, resumable SSE,
43
- session recovery, legacy negotiation, PKCE/resource binding, refresh rotation,
44
- and network-policy enforcement
45
- - **Advanced governed contracts** for cross-model Oracle review, commit-bound
46
- source research, portable Recipes, Git mutation restore, plugin signatures
47
- and relevance, MCP Apps, and authenticated hosted channels
48
- - Dual ESM/CommonJS package conditions for every public entry point, TypeScript
49
- 7 native typechecking, TypeScript 6 compiler-API compatibility, tsup build
50
- - Production dependencies: none
51
-
52
- ### Agent Profiles
53
-
54
- The SDK includes versioned specialist contracts that bind prompts to
55
- enforceable tool, permission, skill, hook, memory, Soul, isolation,
56
- external-action, and completion policies. Profiles compile through monotonic
57
- capability intersection, so organization and host boundaries can narrow but
58
- never widen authority. See [docs/agent-profiles.md](docs/agent-profiles.md).
59
-
60
- ### Artifact Protocol and secure execution foundations
61
-
62
- `@xenosystem/agent-sdk/artifacts` provides the durable interchange
63
- contract for plans, patches, diffs, documents, diagrams, screenshots,
64
- recordings, tests, reviews, releases, SBOMs, provenance, and attestations.
65
- Content revisions are immutable; comments and decisions are append-only; an
66
- approval applies only to the exact revision and content hash reviewed.
67
-
68
- The artifact barrel also provides first-class Wave 3 planning and review
69
- contracts. `XenoSpecLifecycleService` persists separately hashed requirements,
70
- design, task-graph, and plan artifacts; approved plans remain immutable while
71
- execution and task evidence advance through separate revisions.
72
- `detectXenoSpecDrift()` checks plan binding, dependencies, acceptance evidence,
73
- and observed paths. `XenoMultiAgentReviewCoordinator` runs independent review
74
- dimensions, deduplicates findings, requires verifier reproduction (or an
75
- explicit evidence quorum) before marking a finding verified, and emits one
76
- portable `xeno.review-report.v1` artifact.
77
-
78
- `@xenosystem/agent-sdk/intelligence` provides the persistent
79
- repository-index contract: bounded documents and chunks, symbols, imports,
80
- test/document links, Git provenance, deterministic lexical retrieval,
81
- host-supplied exact-model semantic vectors, multi-workspace identity, freshness
82
- inspection, relationship traversal, and an atomic owner-private file store.
83
- Semantic vectors are compared only when model identity and dimensions match.
84
- Hosts retain source collection and embedding-provider authority.
85
-
86
- `@xenosystem/agent-sdk/control-room` projects durable product state
87
- into one hash-bound supervision snapshot. It validates agent hierarchies and
88
- task dependencies, categorizes health and stale heartbeats, aggregates usage,
89
- prioritizes attention, and plans scoped actions. Source adapters can explicitly
90
- narrow advertised actions; they cannot grant controls disallowed by runtime
91
- status. This lets CLI, Hub, and hosted clients share one view without
92
- fabricating unsupported buttons or exposing private reasoning.
93
-
94
- `@xenosystem/agent-sdk/hosted` defines the immutable hosted-
95
- environment, trigger, replay, execution-adapter, cross-device control, and run-
96
- result contracts. Protocol-v2 adapter certificates bind interactive control,
97
- Artifact Protocol output, and Git-result reporting in addition to the OS
98
- containment claims. Hosted messages and exact-scope approval decisions are
99
- normalized, bounded, and tamper-evident; acknowledgements are normalized; run
100
- results hash-bind Git/credential-free PR metadata and artifact/evidence IDs.
101
- These contracts let the Agents API, CLI, and Hub interoperate without treating
102
- policy metadata as proof of containment.
103
-
104
- Graphical and web renderers must import
105
- `@xenosystem/agent-sdk/artifacts/browser`. That browser-safe entry
106
- contains bounded record/diff projection and protocol types without importing
107
- the Node filesystem, process-lock, or crypto implementations owned by the main
108
- artifact barrel.
109
-
110
- The security barrel exposes capability leases and the Secure Execution
111
- Contract. A lease is narrow, expiring, operation-bound authority and cannot
112
- override an Agent Profile or host denial. The execution contract fingerprints
113
- the effective filesystem, network, process, environment, secret,
114
- external-action, browser, and computer-use boundary. These are shared protocol
115
- foundations; they do not by themselves make an OS adapter release-certified.
116
-
117
- ### Governed browser and computer automation
118
-
119
- `@xenosystem/agent-sdk/automation` defines one versioned operation
120
- catalog and adapter contract for XENO Browser and XENO Use. The governed
121
- executor validates the execution-contract fingerprint and real preflight
122
- target, persists required before evidence, consumes an exact capability lease,
123
- dispatches through the adapter, revalidates the resulting target, and persists
124
- after/recording evidence as Artifact Protocol records. Missing required evidence
125
- is quarantined as `evidence-incomplete` rather than accepted as success.
126
-
127
- `XenoLoopbackAutomationAdapter` provides the bearer-authenticated, response-
128
- bounded local transport to XENO Browser. Read and action domain authority are
129
- separate; explicit browser/network deny rules take precedence; operation IDs
130
- are idempotent; and immediate stop propagates to the bound native target. See
131
- [Governed Automation](docs/GOVERNED_AUTOMATION.md).
132
-
133
- ### Agent Skills v2
134
-
135
- `@xenosystem/agent-sdk/skills` provides a host-neutral implementation
136
- of directory-based `SKILL.md` packages. Discovery reads only metadata and hashed
137
- resource descriptors. `createXenoSkillTool()` demand-loads and re-verifies full
138
- instructions when invoked, then intersects tool and external-action policy with
139
- the active host/Profile boundary. Standard `allowed-tools` is treated as a
140
- preapproval request, never as an authority grant. Legacy inline XENO skills can
141
- be adapted without breaking existing hosts.
142
-
143
- ### MCP Streamable HTTP and OAuth
144
-
145
- `@xenosystem/agent-sdk/mcp` targets MCP protocol `2025-11-25`.
146
- URL-only server configs use Streamable HTTP; explicit `sse` remains available
147
- for deprecated servers and can be negotiated only after a rejected Streamable
148
- HTTP initialization. The transport handles protocol/session headers, JSON and
149
- SSE responses, `Last-Event-ID` resumption, duplicate suppression, session
150
- reinitialization, server instructions, manual redirect checks, and permission-
151
- profile validation.
152
-
153
- `MCPOAuthClient` implements protected-resource and authorization-server
154
- discovery, PKCE S256, RFC 8707 resource binding, pre-registered/CIMD/explicit
155
- DCR client selection, authorization-code exchange, refresh-token rotation, and
156
- bounded metadata responses. Hosts own user interaction and secret persistence;
157
- token values are never added to MCP state or model context.
158
-
159
- ### Advanced runtime contracts
160
-
161
- The SDK also exposes additive contracts for cross-model Oracle reports,
162
- commit-bound remote-source research, portable Recipes, automatic Git mutation
163
- checkpoints, signed/locked plugins, repository-aware relevance, MCP Apps, and
164
- authenticated GitLab/Teams/Jira hosted events. These contracts bind identity,
165
- authority, evidence, and persistence without granting a host UI, network,
166
- credential, or deployment authority. See
167
- [Advanced Runtime Contracts](docs/ADVANCED_RUNTIME_CONTRACTS.md).
168
-
169
- ## Delivery automation
170
-
171
- The SDK repo now carries the same production automation discipline as the CLI:
172
-
173
- - `.github/workflows/ci.yml` -- cross-platform build verification plus full Linux runtime checks
174
- - `.github/workflows/certification.yml` -- scheduled production/stress verification and large-repo coding benchmarks
175
- - `.github/workflows/release.yml` -- npm Trusted Publishing with provenance, checksums, and GitHub release artifacts
176
-
177
- ## Architecture
178
-
179
- ```
180
- xeno-agent-sdk/
181
- src/
182
- core/, runtime/, providers/ Agent loop, streaming, context, providers
183
- tools/, terminal/, media/ Tools and managed operations
184
- security/, governance/ Policy, permissions, leases, execution law
185
- session/, memory/, identity/ Durable user and conversation context
186
- orchestration/, control-plane/ Agents, teams, workflows, goals, recovery
187
- agents/, prompts/, soul/ Profiles, prompts, learned state
188
- artifacts/ Artifacts, reviews, provenance, evidence
189
- automation/ Governed browser/computer adapter protocol
190
- oracle/, research/, recipes/ Independent review, source evidence, reusable DAGs
191
- hosted/, intelligence/ Fleet contracts and repository intelligence
192
- skills/ Skills v2 discovery, loading, policy, audit
193
- hooks/, plugins/, mcp/ Extension and interoperability contracts
194
- app-server/, integrations/ Remote and host protocols
195
- audit/, observability/ Audit, usage, telemetry, diagnostics
196
- ui/, electron/, adapters/ Optional UI and host integration layers
197
- persistence/, config/, utils/ Durable and cross-cutting utilities
198
- ```
199
-
200
- ## Selected subsystems
201
-
202
- | Subsystem | Purpose | Key exports |
203
- |-----------|---------|-------------|
204
- | **Core Loop** | Agentic turn loop with tool dispatch, streaming, reducer | `AgentLoop`, `AgentLoopConfig` |
205
- | **Tools** | Demand-loaded registry, built-ins, managed operations, host tools | `ToolRegistry`, `registry` |
206
- | **Security** | Permissions, policy, leases, execution contracts, adapter claims | `PermissionEngine`, `SecureExecutionContract` |
207
- | **Session** | Persistence, checkpoints, transcripts, locks, recovery | `SessionManager`, `CheckpointManager` |
208
- | **Identity** | Persona loading and resolution across 4 levels | `IdentityLoader`, `IdentityResolver` |
209
- | **Memory** | Hierarchical memory with budget management | `MemoryManager`, `MemoryBudget` |
210
- | **Audit** | JSON-lines logging for every action | `AuditLogger` |
211
- | **Config** | Model settings, API endpoints, defaults | `XENO_API_BASE`, `DEFAULT_MODEL` |
212
- | **Delegation** | Planner/executor/reviewer sub-agent workflows | `DelegatedAgent`, `DelegatedTurn` |
213
- | **Artifacts** | Immutable revisions, review lifecycle, anchors, evidence graph | `XenoArtifact`, `ArtifactRepository` |
214
- | **Agent Profiles** | Enforceable specialist runtime identities | `CompiledAgentProfile` |
215
- | **Agent Skills** | Metadata-first `SKILL.md` catalog and demand-loaded invocation | `discoverXenoSkills`, `createXenoSkillTool` |
216
- | **MCP** | Stdio, Streamable HTTP, legacy SSE, OAuth discovery/PKCE/refresh | `MCPManager`, `StreamableHTTPTransport`, `MCPOAuthClient` |
217
- | **Oracle** | Independent opinions, adjudication, citations, disagreement | `XenoOracleCoordinator` |
218
- | **Research** | Commit-bound source excerpts, findings, provenance | `XenoSourceResearchReport` |
219
- | **Recipes** | Strict portable DAGs with authority, budgets, and fingerprints | `compileXenoRecipe` |
220
- | **Plugin trust** | Signatures, locks, publisher policy, advisory relevance | `verifyPluginSupplyChain`, `scorePluginRelevance` |
221
- | **Types** | All shared TypeScript types | `ExecutionMode`, `ResolvedIdentity`, etc. |
222
-
223
- ## How Apps Integrate
224
-
225
- The primary entry point is `createXenoAgent()`. Each app registers its own operations as tools, and the SDK handles everything else: the agentic loop, permission checks, audit logging, memory, sessions.
226
-
227
- ```typescript
228
- import { createXenoAgent } from '@xenosystem/agent-sdk'
229
-
230
- // Each app defines its domain-specific tools
231
- const pixelTools = {
232
- 'layer.remove-bg': {
233
- description: 'Remove background from the active layer',
234
- execute: async (params) => {
235
- const result = await xenoLib.rmbg(params.layerId) // xeno-lib AI model
236
- await engine.applyMask(params.layerId, result.mask) // app engine
237
- return { success: true, layerId: params.layerId }
238
- },
239
- confirm: true, // requires user approval
240
- },
241
- 'brush.draw': {
242
- description: 'Draw a stroke on the active layer',
243
- execute: async (params) => engine.drawStroke(params),
244
- confirm: false, // safe, no confirmation needed
245
- },
246
- 'file.export': {
247
- description: 'Export the current document',
248
- execute: async (params) => exporter.save(params),
249
- confirm: true,
250
- destructive: false,
251
- },
252
- }
253
-
254
- const agent = await createXenoAgent({
255
- toolRegistry: pixelTools,
256
- model: 'claude-sonnet-4-20250514',
257
- permissionConfig: { mode: 'default' },
258
- // identity, memory, session all auto-configured
259
- })
260
- ```
261
-
262
- ### End-to-End Example
263
-
264
- User types in Pixel's agent sidebar: **"Remove the background from this layer"**
265
-
266
- ```
267
- 1. Agent receives natural language input
268
- 2. LLM decides to call tool: layer.remove-bg({ layerId: 'layer-3' })
269
- 3. Permission engine checks: confirm=true -> prompts user for approval
270
- 4. User approves -> tool executes:
271
- a. Calls xeno-lib's RMBG model (Rust, ONNX, GPU-accelerated)
272
- b. Receives alpha mask
273
- c. Applies mask to layer in Pixel's rendering engine
274
- 5. Audit logger records: who=user, what=layer.remove-bg, when=timestamp, result=success
275
- 6. Agent responds: "Done. Background removed from Layer 3."
276
- ```
277
-
278
- ## The Tool Registry Pattern
279
-
280
- Apps extend the SDK by registering their operations as tools. The SDK provides
281
- file, search, shell, background-process, terminal, media, and other reusable
282
- tool contracts; apps add namespaced domain-specific tools:
283
-
284
- | App | Example tools |
285
- |-----|---------------|
286
- | **Pixel** | `brush.draw`, `layer.remove-bg`, `selection.expand`, `filter.blur`, `file.export` |
287
- | **Motion** | `timeline.cut`, `clip.speed`, `transition.add`, `keyframe.set`, `render.export` |
288
- | **Sound** | `track.eq`, `region.normalize`, `master.lufs`, `effect.reverb`, `bounce.export` |
289
- | **Hub** | `workspace.create`, `app.launch`, `agent.dispatch` |
290
-
291
- Each tool declares: `description` (for the LLM), `execute` (the implementation), `confirm` (whether to ask the user), and `destructive` (whether it's irreversible).
292
-
293
- ## Cross-App Orchestration
294
-
295
- One agent request can trigger coordinated work across multiple apps. The Hub acts as an orchestrator, dispatching tasks to per-app agents via a mailbox system.
296
-
297
- ```
298
- User: "Create a product video from these photos with background music"
299
-
300
- xeno-hub (orchestrator)
301
- |-- mailbox.send('pixel-agent', { task: 'export hero images as PNGs' })
302
- |-- mailbox.send('motion-agent', { task: 'create timeline from exported images' })
303
- |-- mailbox.send('sound-agent', { task: 'add background track, master to -14 LUFS' })
304
-
305
- Each agent receives via:
306
- mailbox.onMessage((msg) => agent.execute(msg.task))
307
- ```
308
-
309
- Messages are JSON-serializable for cross-process IPC. The SDK includes agent
310
- protocol/registry and `CrossAppRouter` primitives with bounded delivery.
311
- Production transport, durable replay, authentication, and product adoption are
312
- host responsibilities and must be verified in each consuming repository.
313
-
314
- ## Security Model
315
-
316
- ### Permission Modes
317
-
318
- | Mode | Behavior |
319
- |------|----------|
320
- | `default` | Ask user before writes and shell commands |
321
- | `acceptEdits` | Auto-approve file edits, ask for shell commands |
322
- | `bypassPermissions` | Auto-approve everything (development/testing only) |
323
- | `plan` | Read-only mode, agent can only plan and suggest |
324
-
325
- ### Safety Features
326
-
327
- - **Policy enforcement**: canonical physical-path checks reject traversal through symlinks/junctions, UNC/device, extended-length, drive-relative, root-relative, alternate-data-stream, reserved-device, and ambiguous paths
328
- - **Shell policy**: parsed PowerShell/Bash file operands, redirections, and path parameters enforce read/write capabilities; unresolved dynamic filesystem operands fail closed
329
- - **Execution security levels**: `policy-only`, `process-hardened`, and `contained` are separate API contracts. The legacy `requireOsContainment: true` flag now means strict `contained` and never accepts process hardening as a substitute.
330
- - **Windows process hardening**: `executionLevel: "process-hardened"` uses a reduced primary token, low integrity for read-only execution, an explicit inherited-handle allowlist, a minimized environment, and a kill-on-close Job Object. It does not isolate filesystem reads or network access and is for trusted workspaces only.
331
- - **Fail-closed containment**: the SDK ships five reviewed, digest-pinned runtime executables plus two Windows host-preparation helpers extracted from the integrity-pinned MXC 0.7.0 artifact; it does not install or execute MXC's JavaScript wrapper or transitive `node-pty` graph. Windows x64/ARM64, Linux x64/ARM64, and macOS ARM64 are supported, but asset presence is not certification. Windows setup is an explicit elevated host operation and is never attempted by normal agent execution. `contained` activates only with a short-lived, candidate-, platform-, adapter-, native-report-, and independently signed reviewer-bound host manifest. A source build, unsupported target, mismatch, or missing binding fails with `CONTAINMENT_UNAVAILABLE`; executable discovery is never treated as certification.
332
- - **Destructive action gates**: delete, overwrite, and shell commands require explicit approval
333
- - **Audit trail**: every tool call logged in JSON-lines format with trace IDs
334
- - **Risk classification**: each tool call classified as low/medium/high risk
335
- - **Permission rules**: per-tool, per-path override rules
336
-
337
- The policy layer and Windows process hardening are defense in depth, not a claim that arbitrary untrusted code is contained. See [Security Boundaries](docs/SECURITY_BOUNDARIES.md) for the capability matrix and migration contract.
338
-
339
- ### Managed Shell Processes
340
-
341
- `UnifiedExecManager` is the canonical owner for pipe and PTY process lifecycles. `BackgroundProcessManager` remains the task-oriented compatibility facade and adds owner scoping, bounded file-backed output, same-process foreground-to-background promotion, completion events, input, resize, and human-writer leases.
342
-
343
- - `Bash` accepts `tty: true` only with `run_in_background: true`.
344
- - `TaskInput` writes to or waits on an interactive task. Input text is projected to byte-count/task metadata before audit, permission, callback, and transcript sinks.
345
- - PTY support is adapter based. Hosts register an optional `XenoPtyAdapter`; missing capability returns `PTY_UNAVAILABLE` and never silently falls back to a pipe.
346
- - App-server `exec.start` accepts `tty`, `cols`, and `rows`; `exec.resize` is owner scoped alongside write, output, list, and terminate.
347
- - `SessionManager.recordDirectShellResult` persists bounded local-command output as user-role untrusted context with semantic metadata, not as a fabricated model tool call.
348
-
349
- ### Managed Tool Operations
350
-
351
- Tool calls now receive a stable operation ID, deadline, completion policy (`await`, `observe`, or `detach`), progress counters, terminal reason, and artifact evidence. Recovery snapshots are written atomically under the user control-plane directory, while redacted operation transitions are appended to a private JSONL ledger. Raw command text is not copied into operation lists or the ledger.
352
-
353
- - A promotable foreground `Bash` command moves to the background by changing presentation on the same process. PID, task/process IDs, output cursor, and deadline are retained; the command is never re-executed.
354
- - `TaskOutput({ task_id, offset })` remains a nonblocking compatibility read. `wait: true` adds an event-driven wait capped at 30 seconds and always reports explicit `Operation terminal: yes|no`, state, runtime, idle time, output delta, UTF-8-safe next offset, and suggested action.
355
- - Finite model-created background work defaults to `await`. The parent turn suspends without another provider request and resumes from a durable continuation notification after terminal output arrives. User stop, detach, clear, interrupt, or rewind cancels stale automatic continuation.
356
- - `expected_outputs` on `Bash` records generic file postconditions. Missing or unstable outputs fail the logical operation even after exit code 0; partial artifacts are still reported after timeout/failure.
357
- - `ReadImage` preflights format, dimensions, and bytes. The dependency-free `xeno-owned-raster` path performs owned PNG/DEFLATE decode and encode, JPEG, GIF, VP8/VP8L WebP decode, bounded SVG software rasterization, and resizing; BMP and Netpbm inputs normalize through the same owned PNG path. Oversized supported inputs become owner-private bounded PNG previews without an ambient decoder or canvas fallback. Private previews are removed during session cleanup and originals stay unchanged. The exact approved profiles and explicit rejections are documented in [`docs/owned-media-matrix.md`](docs/owned-media-matrix.md).
358
-
359
- Temporary rollback controls default to enabled: `XENO_TOOL_OPERATION_RUNTIME`, `XENO_AUTO_BACKGROUND_FOREGROUND_BASH`, `XENO_TASK_OUTPUT_WAIT`, `XENO_AUTO_TOOL_CONTINUATION`, and `XENO_READ_IMAGE_AUTO_PREVIEW`. Setting a control to `0` disables its optional behavior without deleting active tasks or persisted operation records.
360
-
361
- On Windows, ordinary managed pipe tasks use the existing kill-on-close Job Object launcher when available and report `windows-job-object`; the explicit `windows-tree-fallback` uses tree termination but is not reported as verified Job ownership. Persistent PowerShell defaults to Windows PowerShell 5.1. Set `XENO_POWERSHELL_EXECUTABLE=pwsh.exe` to use PowerShell 7. Process ownership and cleanup are not claims of untrusted-code containment.
362
-
363
- See [docs/managed-tool-operations.md](docs/managed-tool-operations.md) for the runtime and troubleshooting contract.
364
-
365
- ### Agent Harness Reliability
366
-
367
- - Tool inputs are JSON-schema validated by default; malformed calls are returned to the model as typed errors.
368
- - Interrupted or orphaned tool exchanges are repaired before resumable history is sent back to a provider.
369
- - Repeated identical failing tool calls are circuit-broken before they can loop indefinitely.
370
- - Query lifecycle transitions are emitted as typed runtime events and guarded by idle, active-operation lease, and hard timeouts.
371
- - Context overflow triggers one semantic compact-and-retry path. Old tool payloads are compacted before conversation history is dropped, message count has a hard cap, and preflight token estimates are model-aware with an injectable exact-tokenizer seam.
372
- - The default tool registry demand-loads optional schemas through `ToolSearch` instead of advertising every full schema on every request.
373
-
374
- ### Direct Provider Adapters
375
-
376
- Provider-normalized deployments can continue using the default Xeno API transport. Direct integrations are available from the provider subpath:
377
-
378
- ```typescript
379
- import { createDirectProvider } from '@xenosystem/agent-sdk/providers'
380
- import { createXenoAgent } from '@xenosystem/agent-sdk'
381
-
382
- const provider = createDirectProvider({
383
- kind: 'openai-responses', // 'anthropic' | 'openai-responses' | 'google' | 'openai-compatible'
384
- apiKey: process.env.PROVIDER_API_KEY!,
385
- })
386
-
387
- const { agent } = await createXenoAgent({ provider })
388
- ```
389
-
390
- The adapters translate streaming text, images, parallel tool calls/results, usage, cancellation, and typed failures for Messages, Responses, Gemini, and explicitly profiled chat-compatible endpoints. Generic endpoints require a capability profile and local HTTP requires an explicit local-network endpoint profile. See [docs/PROVIDER_CONFORMANCE.md](docs/PROVIDER_CONFORMANCE.md).
391
-
392
- The provider subpath also exposes a capability-driven product catalog, atomic workspace connection store, bounded credential/model probe, and routing policy. Connections hold credential reference names, never raw credentials. Presets for Azure OpenAI, Bedrock, and Vertex truthfully report `requires-host-adapter` until dedicated adapters are supplied.
393
-
394
- ```typescript
395
- import {
396
- FileXenoProviderConnectionStore,
397
- XenoProviderCatalog,
398
- probeXenoProvider,
399
- selectXenoProviderRoute,
400
- } from '@xenosystem/agent-sdk/providers'
401
- ```
402
-
403
- ### Secure Sharing and Handoff
404
-
405
- `@xenosystem/agent-sdk/sharing` provides deterministic redaction, Ed25519-signed expiring read-only shares, capability/audience/revocation verification, signed session handoffs, and an atomic checksum-protected local registry. Embedded public keys prove integrity but do not imply issuer trust; consuming products supply their own trust store and must intersect handoff authority with local policy.
406
-
407
- ```typescript
408
- import {
409
- createXenoSecureShare,
410
- verifyXenoSecureShare,
411
- createXenoSessionHandoff,
412
- verifyXenoSessionHandoff,
413
- FileXenoShareRegistry,
414
- } from '@xenosystem/agent-sdk/sharing'
415
- ```
416
-
417
- ## Session Persistence
418
-
419
- - Sessions have unique IDs with embedded timestamps
420
- - Full conversation transcripts written to disk
421
- - Checkpoint system for long-running workflows
422
- - Lock management prevents concurrent access to the same session
423
- - Turn restore for crash recovery (resume mid-conversation)
424
- - Session registry tracks all sessions per project
425
-
426
- ## Memory System
427
-
428
- 4-level hierarchy with automatic capture and budget management:
429
-
430
- | Level | Scope | Persists | Example |
431
- |-------|-------|----------|---------|
432
- | **Conversation** | Current turn | No | "The user just asked about Layer 3" |
433
- | **Session** | Current session | Until session ends | "We're working on the hero image" |
434
- | **Project** | Current project | Yes | "This project uses 300 DPI, CMYK" |
435
- | **Global** | All projects | Yes | "User prefers dark theme, metric units" |
436
-
437
- Memory is injected into the system prompt with configurable token budgets to avoid context overflow.
438
-
439
- ## LLM Providers
440
-
441
- | Provider | How | When |
442
- |----------|-----|------|
443
- | **Xeno API** (cloud) | `api.xenostudio.ai` proxy to hosted models | Online, highest capability |
444
- | **xeno-rt** (local) | OpenAI-compatible API on localhost | Offline, privacy, no cost |
445
- | **Ollama** (local) | OpenAI-compatible API | Alternative local provider |
446
-
447
- The SDK is provider-agnostic. OpenAI-compatible endpoints are supported through an explicit capability and endpoint-security profile; compatibility is verified by conformance fixtures rather than inferred from the endpoint label. Provider selection and fallback logic are handled by the config subsystem.
448
-
449
- ## Consumers
450
-
451
- | App | How it uses the SDK | Example agent task |
452
- |-----|--------------------|--------------------|
453
- | **xeno-agent-cli** | Terminal agent (reference implementation) | "Refactor this codebase" |
454
- | **xeno-pixel** | Image editing agent sidebar | "Remove backgrounds from 50 images" |
455
- | **xeno-motion** | Video editing agent sidebar | "Cut this interview into a highlight reel" |
456
- | **xeno-sound** | Audio editing agent sidebar | "Master this podcast to -16 LUFS" |
457
- | **xeno-hub** | Orchestrator routing tasks between apps | "Create marketing materials" dispatches to Pixel + Motion + Sound |
458
-
459
- ## Xeno-Owned UI
460
-
461
- The `/ui` subpath provides a framework-neutral controller, semantic accessibility view, and optional owned DOM renderer. It has no React runtime or peer dependency and remains isolated from the Node-only core entry point.
462
-
463
- ```ts
464
- import {
465
- createAgentUiController,
466
- mountAgentUi,
467
- } from "@xenosystem/agent-sdk/ui";
468
-
469
- const controller = createAgentUiController({
470
- agent: myAgentLoop,
471
- enablePermissionRequests: true,
472
- });
473
- const view = mountAgentUi(document.querySelector("#agent")!, controller, {
474
- agentName: "Xeno Agent",
475
- });
476
-
477
- // App shutdown:
478
- view.dispose();
479
- controller.dispose();
480
- ```
481
-
482
- Framework integrations can subscribe to immutable controller snapshots and call `createAgentUiView` instead of mounting the DOM renderer. Renderer contract version `1` includes message streaming, tool state, token usage, cancellation, retry, permission decisions, safe markdown segments, keyboard submission, and accessible roles/live regions. See [the UI v1 migration guide](docs/migrations/sdk-ui-v1.md).
483
-
484
- ## Ecosystem Position
485
-
486
- ```
487
- LAYER 5 -- APPS (Pixel, Motion, Sound, Hub)
488
- | embed xeno-agent-sdk for AI automation
489
- | agent sidebar in every app
490
- LAYER 3 -- THIS REPO (xeno-agent-sdk)
491
- | uses LLM providers + invokes AI models
492
- LAYER 2 -- COMPUTE (xeno-rt for LLM, xeno-lib for 17 AI models)
493
- | runs on
494
- LAYER 1 -- PLATFORM (servers, auth, credits)
495
- ```
496
-
497
- See [Full Ecosystem Report](../XENO%20CORPORATION%20-%20Full%20Ecosystem%20Report.md) for complete context.
498
-
499
- ## Development
500
-
501
- ```bash
502
- npm install
503
- npm run build # tsup -> ESM + CommonJS package outputs
504
- npm run typecheck # TypeScript 7 native strict-mode check
505
- npm run typecheck:compat # TypeScript 6 compatibility check for API-based tooling
506
- ```
507
-
508
- ## License
509
-
510
- Proprietary and confidential. Copyright (c) 2026 XENO Corporation.
511
- All rights reserved.
1
+ # XENO Agent SDK
2
+
3
+ Core runtime for XENO agent-enabled products. It provides the agent loop, tool execution, permissions, sessions, memory, audit logging, and provider integration used across the XENO platform.
4
+
5
+ ## Runtime Ownership
6
+
7
+ The ownership program targets P2 feature parity, but this checkout must be described by its generated audit rather than by the target. Run `npm run build && npm run compliance:ownership:artifacts` or inspect `docs/compliance/ownership-inventory.md`. Node remains an explicitly external host through P2, and development/build packages remain third-party build inputs. Do not claim that a release is fully proprietary, entirely Xeno-owned, or free of third-party runtime code unless the packaged release smoke prints the corresponding achieved level and the required provenance/counsel gates are complete.
8
+
9
+ The canonical policy and engineering clean-room controls are under `ownership/`.
10
+
11
+ ## Vision
12
+
13
+ Every XENO creative app (Pixel, Motion, Sound) has an AI agent embedded directly into the interface. Users open a sidebar, type what they want ("remove the background from this layer", "cut the silence from this podcast", "match-cut these two clips"), and the agent translates that into tool calls against the app's engine. One request can span multiple apps: "Create a product video from these photos with background music" triggers coordinated work across Pixel, Motion, and Sound simultaneously.
14
+
15
+ The SDK is designed so the same governed agent contracts can be embedded across
16
+ creative products instead of rebuilding orchestration in every application.
17
+
18
+ ```
19
+ Without agent SDK: User manually opens Pixel, edits, opens Motion, edits, opens Sound, edits
20
+ With agent SDK: User says "Create a product video" -> agents in all 3 apps coordinate automatically
21
+ ```
22
+
23
+ ## Current State
24
+
25
+ - **Provider-agnostic runtime** with XENO-hosted, local, direct-provider, and
26
+ generic compatible routes
27
+ - **Extensible tool system** with demand-loaded schemas, managed background
28
+ operations, terminal support, and host-registered domain tools
29
+ - **4 permission modes**: `default`, `acceptEdits`, `bypassPermissions`, `plan`
30
+ - **4-level identity hierarchy**: global, project, role, session
31
+ - **4-level memory system** with auto-capture (conversation, session, project, global)
32
+ - **Session persistence** with checkpoints, transcript writing, and crash recovery
33
+ - **Audit logging** in JSON-lines format (who, what, when, result)
34
+ - **Agent orchestration** with profiles, delegation, teams, goals, workflows,
35
+ monitors, schedules, hooks, and durable recovery
36
+ - **Artifact and evidence protocol** with immutable revisions, review events,
37
+ stable anchors, provenance, and evidence graphs
38
+ - **Secure execution contracts and capability leases** that preserve
39
+ profile/host denials and distinguish policy, hardening, and containment
40
+ - **Agent Skills v2** with interoperable `SKILL.md` discovery, progressive
41
+ disclosure, hash-bound resources, policy intersection, and privacy-safe audit
42
+ - **Current MCP transport and OAuth** with Streamable HTTP, resumable SSE,
43
+ session recovery, legacy negotiation, PKCE/resource binding, refresh rotation,
44
+ and network-policy enforcement
45
+ - **Advanced governed contracts** for cross-model Oracle review, commit-bound
46
+ source research, portable Recipes, Git mutation restore, plugin signatures
47
+ and relevance, MCP Apps, and authenticated hosted channels
48
+ - Dual ESM/CommonJS package conditions for every public entry point, TypeScript
49
+ 7 native typechecking, TypeScript 6 compiler-API compatibility, tsup build
50
+ - Production dependencies: none
51
+
52
+ ### Agent Profiles
53
+
54
+ The SDK includes versioned specialist contracts that bind prompts to
55
+ enforceable tool, permission, skill, hook, memory, Soul, isolation,
56
+ external-action, and completion policies. Profiles compile through monotonic
57
+ capability intersection, so organization and host boundaries can narrow but
58
+ never widen authority. See [docs/agent-profiles.md](docs/agent-profiles.md).
59
+
60
+ ### Artifact Protocol and secure execution foundations
61
+
62
+ `@xenosystem/agent-sdk/artifacts` provides the durable interchange
63
+ contract for plans, patches, diffs, documents, diagrams, screenshots,
64
+ recordings, tests, reviews, releases, SBOMs, provenance, and attestations.
65
+ Content revisions are immutable; comments and decisions are append-only; an
66
+ approval applies only to the exact revision and content hash reviewed.
67
+
68
+ The artifact barrel also provides first-class Wave 3 planning and review
69
+ contracts. `XenoSpecLifecycleService` persists separately hashed requirements,
70
+ design, task-graph, and plan artifacts; approved plans remain immutable while
71
+ execution and task evidence advance through separate revisions.
72
+ `detectXenoSpecDrift()` checks plan binding, dependencies, acceptance evidence,
73
+ and observed paths. `XenoMultiAgentReviewCoordinator` runs independent review
74
+ dimensions, deduplicates findings, requires verifier reproduction (or an
75
+ explicit evidence quorum) before marking a finding verified, and emits one
76
+ portable `xeno.review-report.v1` artifact.
77
+
78
+ `@xenosystem/agent-sdk/intelligence` provides the persistent
79
+ repository-index contract: bounded documents and chunks, symbols, imports,
80
+ test/document links, Git provenance, deterministic lexical retrieval,
81
+ host-supplied exact-model semantic vectors, multi-workspace identity, freshness
82
+ inspection, relationship traversal, and an atomic owner-private file store.
83
+ Semantic vectors are compared only when model identity and dimensions match.
84
+ Hosts retain source collection and embedding-provider authority.
85
+
86
+ `@xenosystem/agent-sdk/control-room` projects durable product state
87
+ into one hash-bound supervision snapshot. It validates agent hierarchies and
88
+ task dependencies, categorizes health and stale heartbeats, aggregates usage,
89
+ prioritizes attention, and plans scoped actions. Source adapters can explicitly
90
+ narrow advertised actions; they cannot grant controls disallowed by runtime
91
+ status. This lets CLI, Hub, and hosted clients share one view without
92
+ fabricating unsupported buttons or exposing private reasoning.
93
+
94
+ `@xenosystem/agent-sdk/hosted` defines the immutable hosted-
95
+ environment, trigger, replay, execution-adapter, cross-device control, and run-
96
+ result contracts. Protocol-v2 adapter certificates bind interactive control,
97
+ Artifact Protocol output, and Git-result reporting in addition to the OS
98
+ containment claims. Hosted messages and exact-scope approval decisions are
99
+ normalized, bounded, and tamper-evident; acknowledgements are normalized; run
100
+ results hash-bind Git/credential-free PR metadata and artifact/evidence IDs.
101
+ These contracts let the Agents API, CLI, and Hub interoperate without treating
102
+ policy metadata as proof of containment.
103
+
104
+ Graphical and web renderers must import
105
+ `@xenosystem/agent-sdk/artifacts/browser`. That browser-safe entry
106
+ contains bounded record/diff projection and protocol types without importing
107
+ the Node filesystem, process-lock, or crypto implementations owned by the main
108
+ artifact barrel.
109
+
110
+ The security barrel exposes capability leases and the Secure Execution
111
+ Contract. A lease is narrow, expiring, operation-bound authority and cannot
112
+ override an Agent Profile or host denial. The execution contract fingerprints
113
+ the effective filesystem, network, process, environment, secret,
114
+ external-action, browser, and computer-use boundary. These are shared protocol
115
+ foundations; they do not by themselves make an OS adapter release-certified.
116
+
117
+ ### Governed browser and computer automation
118
+
119
+ `@xenosystem/agent-sdk/automation` defines one versioned operation
120
+ catalog and adapter contract for XENO Browser and XENO Use. The governed
121
+ executor validates the execution-contract fingerprint and real preflight
122
+ target, persists required before evidence, consumes an exact capability lease,
123
+ dispatches through the adapter, revalidates the resulting target, and persists
124
+ after/recording evidence as Artifact Protocol records. Missing required evidence
125
+ is quarantined as `evidence-incomplete` rather than accepted as success.
126
+
127
+ `XenoLoopbackAutomationAdapter` provides the bearer-authenticated, response-
128
+ bounded local transport to XENO Browser. Read and action domain authority are
129
+ separate; explicit browser/network deny rules take precedence; operation IDs
130
+ are idempotent; and immediate stop propagates to the bound native target. See
131
+ [Governed Automation](docs/GOVERNED_AUTOMATION.md).
132
+
133
+ ### Agent Skills v2
134
+
135
+ `@xenosystem/agent-sdk/skills` provides a host-neutral implementation
136
+ of directory-based `SKILL.md` packages. Discovery reads only metadata and hashed
137
+ resource descriptors. `createXenoSkillTool()` demand-loads and re-verifies full
138
+ instructions when invoked, then intersects tool and external-action policy with
139
+ the active host/Profile boundary. Standard `allowed-tools` is treated as a
140
+ preapproval request, never as an authority grant. Legacy inline XENO skills can
141
+ be adapted without breaking existing hosts.
142
+
143
+ ### MCP Streamable HTTP and OAuth
144
+
145
+ `@xenosystem/agent-sdk/mcp` targets MCP protocol `2025-11-25`.
146
+ URL-only server configs use Streamable HTTP; explicit `sse` remains available
147
+ for deprecated servers and can be negotiated only after a rejected Streamable
148
+ HTTP initialization. The transport handles protocol/session headers, JSON and
149
+ SSE responses, `Last-Event-ID` resumption, duplicate suppression, session
150
+ reinitialization, server instructions, manual redirect checks, and permission-
151
+ profile validation.
152
+
153
+ `MCPOAuthClient` implements protected-resource and authorization-server
154
+ discovery, PKCE S256, RFC 8707 resource binding, pre-registered/CIMD/explicit
155
+ DCR client selection, authorization-code exchange, refresh-token rotation, and
156
+ bounded metadata responses. Hosts own user interaction and secret persistence;
157
+ token values are never added to MCP state or model context.
158
+
159
+ ### Advanced runtime contracts
160
+
161
+ The SDK also exposes additive contracts for cross-model Oracle reports,
162
+ commit-bound remote-source research, portable Recipes, automatic Git mutation
163
+ checkpoints, signed/locked plugins, repository-aware relevance, MCP Apps, and
164
+ authenticated GitLab/Teams/Jira hosted events. These contracts bind identity,
165
+ authority, evidence, and persistence without granting a host UI, network,
166
+ credential, or deployment authority. See
167
+ [Advanced Runtime Contracts](docs/ADVANCED_RUNTIME_CONTRACTS.md).
168
+
169
+ ## Delivery automation
170
+
171
+ The SDK repo now carries the same production automation discipline as the CLI:
172
+
173
+ - `.github/workflows/ci.yml` -- cross-platform build verification plus full Linux runtime checks
174
+ - `.github/workflows/certification.yml` -- scheduled production/stress verification and large-repo coding benchmarks
175
+ - `.github/workflows/release.yml` -- npm Trusted Publishing with provenance, checksums, and GitHub release artifacts
176
+
177
+ ## Architecture
178
+
179
+ ```
180
+ xeno-agent-sdk/
181
+ src/
182
+ core/, runtime/, providers/ Agent loop, streaming, context, providers
183
+ tools/, terminal/, media/ Tools and managed operations
184
+ security/, governance/ Policy, permissions, leases, execution law
185
+ session/, memory/, identity/ Durable user and conversation context
186
+ orchestration/, control-plane/ Agents, teams, workflows, goals, recovery
187
+ agents/, prompts/, soul/ Profiles, prompts, learned state
188
+ artifacts/ Artifacts, reviews, provenance, evidence
189
+ automation/ Governed browser/computer adapter protocol
190
+ oracle/, research/, recipes/ Independent review, source evidence, reusable DAGs
191
+ hosted/, intelligence/ Fleet contracts and repository intelligence
192
+ skills/ Skills v2 discovery, loading, policy, audit
193
+ hooks/, plugins/, mcp/ Extension and interoperability contracts
194
+ app-server/, integrations/ Remote and host protocols
195
+ audit/, observability/ Audit, usage, telemetry, diagnostics
196
+ ui/, electron/, adapters/ Optional UI and host integration layers
197
+ persistence/, config/, utils/ Durable and cross-cutting utilities
198
+ ```
199
+
200
+ ## Selected subsystems
201
+
202
+ | Subsystem | Purpose | Key exports |
203
+ |-----------|---------|-------------|
204
+ | **Core Loop** | Agentic turn loop with tool dispatch, streaming, reducer | `AgentLoop`, `AgentLoopConfig` |
205
+ | **Tools** | Demand-loaded registry, built-ins, managed operations, host tools | `ToolRegistry`, `registry` |
206
+ | **Security** | Permissions, policy, leases, execution contracts, adapter claims | `PermissionEngine`, `SecureExecutionContract` |
207
+ | **Session** | Persistence, checkpoints, transcripts, locks, recovery | `SessionManager`, `CheckpointManager` |
208
+ | **Identity** | Persona loading and resolution across 4 levels | `IdentityLoader`, `IdentityResolver` |
209
+ | **Memory** | Hierarchical memory with budget management | `MemoryManager`, `MemoryBudget` |
210
+ | **Audit** | JSON-lines logging for every action | `AuditLogger` |
211
+ | **Config** | Model settings, API endpoints, defaults | `XENO_API_BASE`, `DEFAULT_MODEL` |
212
+ | **Delegation** | Planner/executor/reviewer sub-agent workflows | `DelegatedAgent`, `DelegatedTurn` |
213
+ | **Artifacts** | Immutable revisions, review lifecycle, anchors, evidence graph | `XenoArtifact`, `ArtifactRepository` |
214
+ | **Agent Profiles** | Enforceable specialist runtime identities | `CompiledAgentProfile` |
215
+ | **Agent Skills** | Metadata-first `SKILL.md` catalog and demand-loaded invocation | `discoverXenoSkills`, `createXenoSkillTool` |
216
+ | **MCP** | Stdio, Streamable HTTP, legacy SSE, OAuth discovery/PKCE/refresh | `MCPManager`, `StreamableHTTPTransport`, `MCPOAuthClient` |
217
+ | **Oracle** | Independent opinions, adjudication, citations, disagreement | `XenoOracleCoordinator` |
218
+ | **Research** | Commit-bound source excerpts, findings, provenance | `XenoSourceResearchReport` |
219
+ | **Recipes** | Strict portable DAGs with authority, budgets, and fingerprints | `compileXenoRecipe` |
220
+ | **Plugin trust** | Signatures, locks, publisher policy, advisory relevance | `verifyPluginSupplyChain`, `scorePluginRelevance` |
221
+ | **Types** | All shared TypeScript types | `ExecutionMode`, `ResolvedIdentity`, etc. |
222
+
223
+ ## How Apps Integrate
224
+
225
+ The primary entry point is `createXenoAgent()`. Each app registers its own operations as tools, and the SDK handles everything else: the agentic loop, permission checks, audit logging, memory, sessions.
226
+
227
+ ```typescript
228
+ import { createXenoAgent } from '@xenosystem/agent-sdk'
229
+
230
+ // Each app defines its domain-specific tools
231
+ const pixelTools = {
232
+ 'layer.remove-bg': {
233
+ description: 'Remove background from the active layer',
234
+ execute: async (params) => {
235
+ const result = await xenoLib.rmbg(params.layerId) // xeno-lib AI model
236
+ await engine.applyMask(params.layerId, result.mask) // app engine
237
+ return { success: true, layerId: params.layerId }
238
+ },
239
+ confirm: true, // requires user approval
240
+ },
241
+ 'brush.draw': {
242
+ description: 'Draw a stroke on the active layer',
243
+ execute: async (params) => engine.drawStroke(params),
244
+ confirm: false, // safe, no confirmation needed
245
+ },
246
+ 'file.export': {
247
+ description: 'Export the current document',
248
+ execute: async (params) => exporter.save(params),
249
+ confirm: true,
250
+ destructive: false,
251
+ },
252
+ }
253
+
254
+ const agent = await createXenoAgent({
255
+ toolRegistry: pixelTools,
256
+ model: 'claude-sonnet-4-20250514',
257
+ permissionConfig: { mode: 'default' },
258
+ // identity, memory, session all auto-configured
259
+ })
260
+ ```
261
+
262
+ ### End-to-End Example
263
+
264
+ User types in Pixel's agent sidebar: **"Remove the background from this layer"**
265
+
266
+ ```
267
+ 1. Agent receives natural language input
268
+ 2. LLM decides to call tool: layer.remove-bg({ layerId: 'layer-3' })
269
+ 3. Permission engine checks: confirm=true -> prompts user for approval
270
+ 4. User approves -> tool executes:
271
+ a. Calls xeno-lib's RMBG model (Rust, ONNX, GPU-accelerated)
272
+ b. Receives alpha mask
273
+ c. Applies mask to layer in Pixel's rendering engine
274
+ 5. Audit logger records: who=user, what=layer.remove-bg, when=timestamp, result=success
275
+ 6. Agent responds: "Done. Background removed from Layer 3."
276
+ ```
277
+
278
+ ## The Tool Registry Pattern
279
+
280
+ Apps extend the SDK by registering their operations as tools. The SDK provides
281
+ file, search, shell, background-process, terminal, media, and other reusable
282
+ tool contracts; apps add namespaced domain-specific tools:
283
+
284
+ | App | Example tools |
285
+ |-----|---------------|
286
+ | **Pixel** | `brush.draw`, `layer.remove-bg`, `selection.expand`, `filter.blur`, `file.export` |
287
+ | **Motion** | `timeline.cut`, `clip.speed`, `transition.add`, `keyframe.set`, `render.export` |
288
+ | **Sound** | `track.eq`, `region.normalize`, `master.lufs`, `effect.reverb`, `bounce.export` |
289
+ | **Hub** | `workspace.create`, `app.launch`, `agent.dispatch` |
290
+
291
+ Each tool declares: `description` (for the LLM), `execute` (the implementation), `confirm` (whether to ask the user), and `destructive` (whether it's irreversible).
292
+
293
+ ## Cross-App Orchestration
294
+
295
+ One agent request can trigger coordinated work across multiple apps. The Hub acts as an orchestrator, dispatching tasks to per-app agents via a mailbox system.
296
+
297
+ ```
298
+ User: "Create a product video from these photos with background music"
299
+
300
+ xeno-hub (orchestrator)
301
+ |-- mailbox.send('pixel-agent', { task: 'export hero images as PNGs' })
302
+ |-- mailbox.send('motion-agent', { task: 'create timeline from exported images' })
303
+ |-- mailbox.send('sound-agent', { task: 'add background track, master to -14 LUFS' })
304
+
305
+ Each agent receives via:
306
+ mailbox.onMessage((msg) => agent.execute(msg.task))
307
+ ```
308
+
309
+ Messages are JSON-serializable for cross-process IPC. The SDK includes agent
310
+ protocol/registry and `CrossAppRouter` primitives with bounded delivery.
311
+ Production transport, durable replay, authentication, and product adoption are
312
+ host responsibilities and must be verified in each consuming repository.
313
+
314
+ ## Security Model
315
+
316
+ ### Permission Modes
317
+
318
+ | Mode | Behavior |
319
+ |------|----------|
320
+ | `default` | Ask user before writes and shell commands |
321
+ | `acceptEdits` | Auto-approve file edits, ask for shell commands |
322
+ | `bypassPermissions` | Auto-approve everything (development/testing only) |
323
+ | `plan` | Read-only mode, agent can only plan and suggest |
324
+
325
+ ### Safety Features
326
+
327
+ - **Policy enforcement**: canonical physical-path checks reject traversal through symlinks/junctions, UNC/device, extended-length, drive-relative, root-relative, alternate-data-stream, reserved-device, and ambiguous paths
328
+ - **Shell policy**: parsed PowerShell/Bash file operands, redirections, and path parameters enforce read/write capabilities; unresolved dynamic filesystem operands fail closed
329
+ - **Execution security levels**: `policy-only`, `process-hardened`, and `contained` are separate API contracts. The legacy `requireOsContainment: true` flag now means strict `contained` and never accepts process hardening as a substitute.
330
+ - **Windows process hardening**: `executionLevel: "process-hardened"` uses a reduced primary token, low integrity for read-only execution, an explicit inherited-handle allowlist, a minimized environment, and a kill-on-close Job Object. It does not isolate filesystem reads or network access and is for trusted workspaces only.
331
+ - **Fail-closed containment**: the SDK ships five reviewed, digest-pinned runtime executables plus two Windows host-preparation helpers extracted from the integrity-pinned MXC 0.7.0 artifact; it does not install or execute MXC's JavaScript wrapper or transitive `node-pty` graph. Windows x64/ARM64, Linux x64/ARM64, and macOS ARM64 are supported, but asset presence is not certification. Windows setup is an explicit elevated host operation and is never attempted by normal agent execution. `contained` activates only with a short-lived, candidate-, platform-, adapter-, native-report-, and independently signed reviewer-bound host manifest. A source build, unsupported target, mismatch, or missing binding fails with `CONTAINMENT_UNAVAILABLE`; executable discovery is never treated as certification.
332
+ - **Destructive action gates**: delete, overwrite, and shell commands require explicit approval
333
+ - **Audit trail**: every tool call logged in JSON-lines format with trace IDs
334
+ - **Risk classification**: each tool call classified as low/medium/high risk
335
+ - **Permission rules**: per-tool, per-path override rules
336
+
337
+ The policy layer and Windows process hardening are defense in depth, not a claim that arbitrary untrusted code is contained. See [Security Boundaries](docs/SECURITY_BOUNDARIES.md) for the capability matrix and migration contract.
338
+
339
+ ### Managed Shell Processes
340
+
341
+ `UnifiedExecManager` is the canonical owner for pipe and PTY process lifecycles. `BackgroundProcessManager` remains the task-oriented compatibility facade and adds owner scoping, bounded file-backed output, same-process foreground-to-background promotion, completion events, input, resize, and human-writer leases.
342
+
343
+ - `Bash` accepts `tty: true` only with `run_in_background: true`.
344
+ - `TaskInput` writes to or waits on an interactive task. Input text is projected to byte-count/task metadata before audit, permission, callback, and transcript sinks.
345
+ - PTY support is adapter based. Hosts register an optional `XenoPtyAdapter`; missing capability returns `PTY_UNAVAILABLE` and never silently falls back to a pipe.
346
+ - App-server `exec.start` accepts `tty`, `cols`, and `rows`; `exec.resize` is owner scoped alongside write, output, list, and terminate.
347
+ - `SessionManager.recordDirectShellResult` persists bounded local-command output as user-role untrusted context with semantic metadata, not as a fabricated model tool call.
348
+
349
+ ### Managed Tool Operations
350
+
351
+ Tool calls now receive a stable operation ID, deadline, completion policy (`await`, `observe`, or `detach`), progress counters, terminal reason, and artifact evidence. Recovery snapshots are written atomically under the user control-plane directory, while redacted operation transitions are appended to a private JSONL ledger. Raw command text is not copied into operation lists or the ledger.
352
+
353
+ - A promotable foreground `Bash` command moves to the background by changing presentation on the same process. PID, task/process IDs, output cursor, and deadline are retained; the command is never re-executed.
354
+ - `TaskOutput({ task_id, offset })` remains a nonblocking compatibility read. `wait: true` adds an event-driven wait capped at 30 seconds and always reports explicit `Operation terminal: yes|no`, state, runtime, idle time, output delta, UTF-8-safe next offset, and suggested action.
355
+ - Finite model-created background work defaults to `await`. The parent turn suspends without another provider request and resumes from a durable continuation notification after terminal output arrives. User stop, detach, clear, interrupt, or rewind cancels stale automatic continuation.
356
+ - `expected_outputs` on `Bash` records generic file postconditions. Missing or unstable outputs fail the logical operation even after exit code 0; partial artifacts are still reported after timeout/failure.
357
+ - `ReadImage` preflights format, dimensions, and bytes. The dependency-free `xeno-owned-raster` path performs owned PNG/DEFLATE decode and encode, JPEG, GIF, VP8/VP8L WebP decode, bounded SVG software rasterization, and resizing; BMP and Netpbm inputs normalize through the same owned PNG path. Oversized supported inputs become owner-private bounded PNG previews without an ambient decoder or canvas fallback. Private previews are removed during session cleanup and originals stay unchanged. The exact approved profiles and explicit rejections are documented in [`docs/owned-media-matrix.md`](docs/owned-media-matrix.md).
358
+
359
+ Temporary rollback controls default to enabled: `XENO_TOOL_OPERATION_RUNTIME`, `XENO_AUTO_BACKGROUND_FOREGROUND_BASH`, `XENO_TASK_OUTPUT_WAIT`, `XENO_AUTO_TOOL_CONTINUATION`, and `XENO_READ_IMAGE_AUTO_PREVIEW`. Setting a control to `0` disables its optional behavior without deleting active tasks or persisted operation records.
360
+
361
+ On Windows, ordinary managed pipe tasks use the existing kill-on-close Job Object launcher when available and report `windows-job-object`; the explicit `windows-tree-fallback` uses tree termination but is not reported as verified Job ownership. Persistent PowerShell defaults to Windows PowerShell 5.1. Set `XENO_POWERSHELL_EXECUTABLE=pwsh.exe` to use PowerShell 7. Process ownership and cleanup are not claims of untrusted-code containment.
362
+
363
+ See [docs/managed-tool-operations.md](docs/managed-tool-operations.md) for the runtime and troubleshooting contract.
364
+
365
+ ### Agent Harness Reliability
366
+
367
+ - Tool inputs are JSON-schema validated by default; malformed calls are returned to the model as typed errors.
368
+ - Interrupted or orphaned tool exchanges are repaired before resumable history is sent back to a provider.
369
+ - Repeated identical failing tool calls are circuit-broken before they can loop indefinitely.
370
+ - Query lifecycle transitions are emitted as typed runtime events and guarded by idle, active-operation lease, and hard timeouts.
371
+ - Context overflow triggers one semantic compact-and-retry path. Old tool payloads are compacted before conversation history is dropped, message count has a hard cap, and preflight token estimates are model-aware with an injectable exact-tokenizer seam.
372
+ - The default tool registry demand-loads optional schemas through `ToolSearch` instead of advertising every full schema on every request.
373
+
374
+ ### Direct Provider Adapters
375
+
376
+ Provider-normalized deployments can continue using the default Xeno API transport. Direct integrations are available from the provider subpath:
377
+
378
+ ```typescript
379
+ import { createDirectProvider } from '@xenosystem/agent-sdk/providers'
380
+ import { createXenoAgent } from '@xenosystem/agent-sdk'
381
+
382
+ const provider = createDirectProvider({
383
+ kind: 'openai-responses', // 'anthropic' | 'openai-responses' | 'google' | 'openai-compatible'
384
+ apiKey: process.env.PROVIDER_API_KEY!,
385
+ })
386
+
387
+ const { agent } = await createXenoAgent({ provider })
388
+ ```
389
+
390
+ The adapters translate streaming text, images, parallel tool calls/results, usage, cancellation, and typed failures for Messages, Responses, Gemini, and explicitly profiled chat-compatible endpoints. Generic endpoints require a capability profile and local HTTP requires an explicit local-network endpoint profile. See [docs/PROVIDER_CONFORMANCE.md](docs/PROVIDER_CONFORMANCE.md).
391
+
392
+ The provider subpath also exposes a capability-driven product catalog, atomic workspace connection store, bounded credential/model probe, and routing policy. Connections hold credential reference names, never raw credentials. Presets for Azure OpenAI, Bedrock, and Vertex truthfully report `requires-host-adapter` until dedicated adapters are supplied.
393
+
394
+ ```typescript
395
+ import {
396
+ FileXenoProviderConnectionStore,
397
+ XenoProviderCatalog,
398
+ probeXenoProvider,
399
+ selectXenoProviderRoute,
400
+ } from '@xenosystem/agent-sdk/providers'
401
+ ```
402
+
403
+ ### Secure Sharing and Handoff
404
+
405
+ `@xenosystem/agent-sdk/sharing` provides deterministic redaction, Ed25519-signed expiring read-only shares, capability/audience/revocation verification, signed session handoffs, and an atomic checksum-protected local registry. Embedded public keys prove integrity but do not imply issuer trust; consuming products supply their own trust store and must intersect handoff authority with local policy.
406
+
407
+ ```typescript
408
+ import {
409
+ createXenoSecureShare,
410
+ verifyXenoSecureShare,
411
+ createXenoSessionHandoff,
412
+ verifyXenoSessionHandoff,
413
+ FileXenoShareRegistry,
414
+ } from '@xenosystem/agent-sdk/sharing'
415
+ ```
416
+
417
+ ## Session Persistence
418
+
419
+ - Sessions have unique IDs with embedded timestamps
420
+ - Full conversation transcripts written to disk
421
+ - Checkpoint system for long-running workflows
422
+ - Lock management prevents concurrent access to the same session
423
+ - Turn restore for crash recovery (resume mid-conversation)
424
+ - Session registry tracks all sessions per project
425
+
426
+ ## Memory System
427
+
428
+ 4-level hierarchy with automatic capture and budget management:
429
+
430
+ | Level | Scope | Persists | Example |
431
+ |-------|-------|----------|---------|
432
+ | **Conversation** | Current turn | No | "The user just asked about Layer 3" |
433
+ | **Session** | Current session | Until session ends | "We're working on the hero image" |
434
+ | **Project** | Current project | Yes | "This project uses 300 DPI, CMYK" |
435
+ | **Global** | All projects | Yes | "User prefers dark theme, metric units" |
436
+
437
+ Memory is injected into the system prompt with configurable token budgets to avoid context overflow.
438
+
439
+ ## LLM Providers
440
+
441
+ | Provider | How | When |
442
+ |----------|-----|------|
443
+ | **Xeno API** (cloud) | `api.xenostudio.ai` proxy to hosted models | Online, highest capability |
444
+ | **xeno-rt** (local) | OpenAI-compatible API on localhost | Offline, privacy, no cost |
445
+ | **Ollama** (local) | OpenAI-compatible API | Alternative local provider |
446
+
447
+ The SDK is provider-agnostic. OpenAI-compatible endpoints are supported through an explicit capability and endpoint-security profile; compatibility is verified by conformance fixtures rather than inferred from the endpoint label. Provider selection and fallback logic are handled by the config subsystem.
448
+
449
+ ## Consumers
450
+
451
+ | App | How it uses the SDK | Example agent task |
452
+ |-----|--------------------|--------------------|
453
+ | **xeno-agent-cli** | Terminal agent (reference implementation) | "Refactor this codebase" |
454
+ | **xeno-pixel** | Image editing agent sidebar | "Remove backgrounds from 50 images" |
455
+ | **xeno-motion** | Video editing agent sidebar | "Cut this interview into a highlight reel" |
456
+ | **xeno-sound** | Audio editing agent sidebar | "Master this podcast to -16 LUFS" |
457
+ | **xeno-hub** | Orchestrator routing tasks between apps | "Create marketing materials" dispatches to Pixel + Motion + Sound |
458
+
459
+ ## Xeno-Owned UI
460
+
461
+ The `/ui` subpath provides a framework-neutral controller, semantic accessibility view, and optional owned DOM renderer. It has no React runtime or peer dependency and remains isolated from the Node-only core entry point.
462
+
463
+ ```ts
464
+ import {
465
+ createAgentUiController,
466
+ mountAgentUi,
467
+ } from "@xenosystem/agent-sdk/ui";
468
+
469
+ const controller = createAgentUiController({
470
+ agent: myAgentLoop,
471
+ enablePermissionRequests: true,
472
+ });
473
+ const view = mountAgentUi(document.querySelector("#agent")!, controller, {
474
+ agentName: "Xeno Agent",
475
+ });
476
+
477
+ // App shutdown:
478
+ view.dispose();
479
+ controller.dispose();
480
+ ```
481
+
482
+ Framework integrations can subscribe to immutable controller snapshots and call `createAgentUiView` instead of mounting the DOM renderer. Renderer contract version `1` includes message streaming, tool state, token usage, cancellation, retry, permission decisions, safe markdown segments, keyboard submission, and accessible roles/live regions. See [the UI v1 migration guide](docs/migrations/sdk-ui-v1.md).
483
+
484
+ ## Ecosystem Position
485
+
486
+ ```
487
+ LAYER 5 -- APPS (Pixel, Motion, Sound, Hub)
488
+ | embed xeno-agent-sdk for AI automation
489
+ | agent sidebar in every app
490
+ LAYER 3 -- THIS REPO (xeno-agent-sdk)
491
+ | uses LLM providers + invokes AI models
492
+ LAYER 2 -- COMPUTE (xeno-rt for LLM, xeno-lib for 17 AI models)
493
+ | runs on
494
+ LAYER 1 -- PLATFORM (servers, auth, credits)
495
+ ```
496
+
497
+ See [Full Ecosystem Report](../XENO%20CORPORATION%20-%20Full%20Ecosystem%20Report.md) for complete context.
498
+
499
+ ## Development
500
+
501
+ ```bash
502
+ npm install
503
+ npm run build # tsup -> ESM + CommonJS package outputs
504
+ npm run typecheck # TypeScript 7 native strict-mode check
505
+ npm run typecheck:compat # TypeScript 6 compatibility check for API-based tooling
506
+ ```
507
+
508
+ ## License
509
+
510
+ Proprietary and confidential. Copyright (c) 2026 XENO Corporation.
511
+ All rights reserved.