@staix/agent-hub 0.12.15 → 0.12.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/README.md +18 -5
  3. package/docs/agent-notes/adapters.md +1 -0
  4. package/docs/agent-notes/budget.md +5 -1
  5. package/docs/agent-notes/bus.md +2 -0
  6. package/docs/agent-notes/tasks.md +3 -1
  7. package/docs/agent-notes/tests.md +3 -1
  8. package/docs/operations.md +277 -11
  9. package/docs/quickstart.md +4 -4
  10. package/docs/security.md +26 -1
  11. package/docs/smoke.md +93 -0
  12. package/docs/specs/2026-09-19-agent-hub-design.md +223 -0
  13. package/docs/verification/2026-10-09-agent-shell-t0.md +123 -0
  14. package/docs/verified.json +168 -0
  15. package/package.json +1 -1
  16. package/plugins/agent-hub/.claude-plugin/plugin.json +1 -1
  17. package/plugins/agent-hub/server.js +18 -6
  18. package/src/adapters/acp.ts +2 -2
  19. package/src/adapters/claude-channel.ts +3 -2
  20. package/src/adapters/codex-appserver.ts +10 -2
  21. package/src/adapters/local-worker.ts +5 -5
  22. package/src/adapters/pi.ts +4 -14
  23. package/src/cli/console-state.ts +213 -0
  24. package/src/cli/console.ts +201 -0
  25. package/src/cli/facts-hook.ts +8 -1
  26. package/src/cli/identity-audit.ts +66 -0
  27. package/src/cli/identity.ts +54 -0
  28. package/src/cli/init.ts +40 -30
  29. package/src/cli/launch.ts +31 -1
  30. package/src/cli/main.ts +95 -39
  31. package/src/cli/preview.ts +80 -0
  32. package/src/cli/status-lines.ts +8 -2
  33. package/src/cli/statusline-tee.ts +14 -0
  34. package/src/cli/tail-render.ts +17 -0
  35. package/src/cli/upgrade-runtime.ts +1 -1
  36. package/src/hub/board.ts +9 -2
  37. package/src/hub/bus.ts +82 -10
  38. package/src/hub/child-process.ts +11 -0
  39. package/src/hub/conductor.ts +196 -0
  40. package/src/hub/context-window.ts +72 -0
  41. package/src/hub/control-client.ts +3 -3
  42. package/src/hub/daemon.ts +492 -77
  43. package/src/hub/envelope.ts +1 -1
  44. package/src/hub/events.ts +7 -1
  45. package/src/hub/hub-tools.ts +14 -1
  46. package/src/hub/report.ts +53 -4
  47. package/src/hub/supervision.ts +152 -0
  48. package/src/hub/task-sweep.ts +59 -0
  49. package/src/hub/tasks.ts +86 -15
  50. package/src/hub/usage.ts +37 -2
  51. package/src/models/relay.ts +6 -1
  52. package/src/pi/launch.ts +21 -0
  53. package/src/ui/index.html +6 -0
package/docs/smoke.md CHANGED
@@ -2,6 +2,99 @@
2
2
 
3
3
  `scripts/check.sh` covers everything against fakes. The legs below need real accounts and an interactive terminal, so they are run by hand and recorded here.
4
4
 
5
+ ## Operator console and conductor candidate (#190, #191, #193-#195)
6
+
7
+ Candidate 0.12.17, protocol 15, observed on 2026-10-09 KST. The final Claude
8
+ observer runs used immutable source `20ad5e6`; their native completion receipts
9
+ and actual console effects were checked independently. The Kimi new-tool leg
10
+ remains quota-blocked, as recorded below.
11
+
12
+ - Actual Codex 0.146.0 (`gpt-5.5`) and Claude 2.1.295 (Opus 5.5) TUIs started
13
+ real local and headless Pi peers, proposed exactly two tasks owned initially
14
+ by local, reassigned Beta to Pi, and placed and released their own holds.
15
+ The real owners checked their outputs and called `hub_task_done`; the native
16
+ conductor independently read the exact files before approving both tasks.
17
+ Real CLI calls retained the agent actor, and human-only queue/permission
18
+ actions were refused from an agent shell.
19
+ - The person authorized `alpha.txt = ALPHA` (5 bytes) and `beta.txt = BETA`
20
+ (4 bytes), both without a newline. Separate selection and confirmation keys
21
+ reached the actual console PTY and produced allow-once console audit records.
22
+ Out-of-scope directory-list commands were cancelled. The final Claude own
23
+ run used three allow-once answers because Pi combined its exact write and
24
+ bounded byte checks in one approved command; it also recorded two cancelled
25
+ local directory-list requests. No permission or task completion was fabricated.
26
+ - The final Claude own run recorded nine completed native turns and nine
27
+ matching daemon Stop records, 27 unique native usage records, and
28
+ 1,888,242 tokens including cached input. Eight completed native turns contained
29
+ supervision, with 1,292,922 recorded tokens. Its final message UUID/id,
30
+ `end_turn`, `turn_duration`, instance/session/launcher and opaque Stop receipt
31
+ matched; native idle and an empty pending delivery queue were verified before
32
+ owned-process cleanup. Local recorded three turns and 72,046 gateway tokens;
33
+ Pi recorded two turns and 47,177 native tokens.
34
+ - A read-only Claude feed-off continuation on the final observer preserved the
35
+ existing approved tasks and exact files. It verified one completed native turn,
36
+ four usage records and 263,413 cached-inclusive tokens, with matching daemon
37
+ Stop, native idle and delivery settlement. An obsolete review hold was first
38
+ discarded through the authenticated public operator API after fresh approved
39
+ task/history and exact BETA readback. That discard is not native execution.
40
+ - A real console with two pending cards and no operator input consumed 0.07 CPU
41
+ seconds over 40.018 wall seconds in stream mode (0.175% of one CPU). A separate
42
+ actual Peers panel, 120x40, with two pending cards and no operator input used
43
+ 0.07 CPU seconds over 40.012 wall seconds (0.175%). These are cumulative `ps`
44
+ process CPU samples in distinct modes/runs, not whole-host idle measurements.
45
+ - In a plain Claude TUI without the development-channel flag, one actual
46
+ `hub_status` MCP call succeeded and the daemon audited the Claude status
47
+ action. A unique directed operator push was accepted by the bridge, but no
48
+ pushed native user row or answer appeared in the observed 20.075-second window.
49
+ This proves tool access separately from the bounded negative push observation.
50
+ The real channel-enabled runs above received actual review and supervision
51
+ pushes. The plain probe and cleanup completed in 56.724 seconds, with no model
52
+ file, shell, task or settings actions.
53
+ - Real Kimi Code CLI 2.1.1 completed ACP initialization and `session/new`, then
54
+ rejected the single prompt with HTTP 403 for its weekly account usage limit.
55
+ No requested native tool event or role-refusal result occurred. The reset time
56
+ was not supplied. The owning CLI listed only the managed OAuth Kimi provider
57
+ (four models, default `kimi-code/k3`), so no configured alternative provider was
58
+ found. This is an incomplete external prerequisite, not a new-tool pass. No
59
+ purchase, provider/configuration change or model retry was performed; owned
60
+ processes were stopped after 3.142 seconds.
61
+
62
+ The user explicitly approved release 0.12.17 with only the Kimi native new-tool
63
+ and ordinary-role refusal T0 evidence deferred on 2026-10-09. That native
64
+ prerequisite remains unverified; the quota rejection is not a tool pass. This
65
+ decision does not defer the other native/source gates or authorize a purchase,
66
+ credentials, provider/configuration change or account retry. The same isolated
67
+ read-only Kimi probe remains required before its native coverage is marked verified.
68
+
69
+ Earlier captures remain part of the evidence:
70
+
71
+ - Codex feed-off included an interrupted original and read-only continuation:
72
+ six logical native turns, 49 increments and 1,979,016 tokens. Codex feed-own
73
+ recorded nine turns, 34 increments and 1,399,470 tokens. Their journals were
74
+ independently checked after cleanup. The first baseline interruption and
75
+ operator reconciliation are preserved; neither native task completion nor
76
+ board approval was substituted by the harness.
77
+ - The original Claude runs approved the actual files but an early idle-based
78
+ harness stopped their final review answers. Subsequent `da2ebbc` feed-off
79
+ completed five native turns and 1,451,782 tokens but certified only one daemon
80
+ Stop. A `6034dd9` read-only continuation completed another native turn and
81
+ 261,483 tokens, while its Stop remained unavailable during the native hook.
82
+ Both incomplete observer captures are preserved alongside raw and audited
83
+ summaries. The final post-ACK observer above closes the fresh completion proof;
84
+ it does not retroactively certify the earlier missed Stop records.
85
+
86
+ All token totals describe whole native/model turns, including cached input and
87
+ other work. Operator waiting, cancelled requests, marker probes, interrupted
88
+ continuations and differing source revisions make these observations unsuitable
89
+ for a causal feed-overhead or model-efficiency comparison. Unknown measurements
90
+ and the Kimi prerequisite remain explicit.
91
+
92
+ The harness uses real Python PTYs. Operator-file-input forwards only
93
+ chat-authorized keys, removes inherited Orca terminal ownership from fixture
94
+ children, and generates no approval automatically. Private original and
95
+ continuation captures remain separate. The shell-marker probes and vendor
96
+ limits are recorded in [the identity T0 ledger](verification/2026-10-09-agent-shell-t0.md).
97
+
5
98
  ## Approval race live reproduction and candidate verification (#98)
6
99
 
7
100
  Measured on 2026-10-02 KST with installed 0.12.1 and the correction candidate,
@@ -1242,3 +1242,226 @@ Relay native-session counters and per-dispatch transport usage are separate
1242
1242
  measurements. Request usage and provider availability are optional metadata,
1243
1243
  bound to their own dispatch IDs; provider absence does not change model
1244
1244
  qualification. Primary and fallback dispatches have independent outcomes.
1245
+
1246
+
1247
+ ## Task idle sweep (#186, 2026-10-09)
1248
+
1249
+ `Tasks.sweep(now)` owns a deterministic, default-off between-turn sweep. It
1250
+ classifies proposed tasks with an owner as unaccepted assignments, in-progress
1251
+ tasks with an idle owner as idle-owner findings, and in-review tasks with a
1252
+ reviewer as review-pending findings. Separate minute thresholds and a ladder
1253
+ interval are configured through the strict `task_sweep` config block described
1254
+ in operations. The daemon ticks only an enabled sweep and clears its timer on
1255
+ shutdown. No control message shape changes.
1256
+
1257
+ The activity identity is the last real history entry index, so simultaneous
1258
+ real events are distinct. Typed sweep entries persist finding kind, activity
1259
+ index, step and injected sweep time in the existing hub.db task history. They
1260
+ never reset activity or invalidate a pending completion check. Persisting a
1261
+ step precedes its notice: restart does not repeat the step, but a crash between
1262
+ write and publish can leave the attempted notice unpublished or uncertain.
1263
+ This is not an exactly-once delivery claim.
1264
+
1265
+ The ladder sends one ordinary task reminder to the responsible peer, then
1266
+ notifies the console and available planner-role peers, then suggests an
1267
+ alternative from pure `assign()`. Every notice uses public task titles and
1268
+ numeric task refs only; PII text and task refs/plans are absent from notices and
1269
+ ladder records. Busy/paused/offline/native-active peers, unresolved dependencies,
1270
+ completion checks, queued/in-flight/held deliveries, recovery/shutdown and live
1271
+ silent cohorts suppress the sweep. Human review reminders go to the console.
1272
+
1273
+ The Claude launcher and preview share one native-observation hook selector.
1274
+ Turn-free coordination selects facts observations; an enabled task sweep also
1275
+ selects the existing PreToolUse/PostToolUse/Stop transport in advisory mode.
1276
+ Advisory observations update native turn evidence without enabling facts
1277
+ injection. A delivery receipt or task transition never counts as a native Stop.
1278
+ Explicit caller settings remain authoritative, with a diagnostic that the
1279
+ managed idle observation hooks are disabled for that session.
1280
+
1281
+ Automatic reassignment remains off. Explicit `auto_reassign: true` enables only
1282
+ an owner handover to an available routed alternative through Tasks' existing
1283
+ assignment path; reviewer handovers remain suggestions. No failed-work outcome
1284
+ is inferred from elapsed time. Existing route-explain behavior is unchanged,
1285
+ and offline-owner release and delivery-journal retries remain separate policies.
1286
+
1287
+ ## Initialization and launcher previews (issue #189)
1288
+
1289
+ Initialization provides a read-only action/path/reason plan with managed-block
1290
+ insert/replace/remove metadata. Normal initialization applies the same planner.
1291
+
1292
+ Claude/Codex/Kimi/Pi launcher previews share native command builders with actual
1293
+ launches. Preview exits before project registration, runtime setup or terminal
1294
+ ownership records. JSON contains argv/settings with arbitrary user-supplied
1295
+ values and custom executable overrides redacted, environment names only, and
1296
+ explicit reasons for unresolved native-assigned endpoints/session identities.
1297
+ A preview does not assert runtime, account or executable readiness.
1298
+ Claude/Codex previews report the conditional daemon/Orca launcher identity
1299
+ environment names and unresolved reasons without reading runtime identity,
1300
+ querying Orca, allocating a launch id or exposing values. Existing Pi launch
1301
+ behavior and its builder-derived environment preview remain unchanged.
1302
+
1303
+
1304
+
1305
+ ## Native context telemetry and checkpoints (issue #185)
1306
+
1307
+ Context occupancy is separate from quota and accumulated billable session usage.
1308
+ Claude reports `context_window.used_percentage` and `context_window_size` from
1309
+ its status-line payload. A valid `current_usage` counter tuple is required;
1310
+ null, malformed or missing current usage is unknown, including immediately
1311
+ after compaction. Its input occupancy excludes output and sums input tokens,
1312
+ cache creation and cache reads, matching the [official status-line schema](https://code.claude.com/docs/en/statusline).
1313
+ Codex reports `tokenUsage.last.totalTokens / tokenUsage.modelContextWindow` in
1314
+ `thread/tokenUsage/updated`. The [native protocol](https://github.com/openai/codex/blob/main/codex-rs/app-server-protocol/schema/typescript/v2/ThreadTokenUsage.ts)
1315
+ and [native TUI](https://github.com/openai/codex/blob/main/codex-rs/tui/src/token_usage.rs)
1316
+ distinguish the last active context from the accumulated `total`; the displayed
1317
+ raw occupancy does not apply the TUI's baseline-adjusted remaining percentage.
1318
+ Pi's RPC state exposes no measured native counter. Its extension context API
1319
+ returns an estimate, so Pi and unsupported peers remain unknown here.
1320
+
1321
+ Each reading records source and measurement time. Claude's status-line file is
1322
+ bound to the daemon instance, managed launcher and native session; Codex's
1323
+ notification is fenced by its owning link and native thread. Stale, invalid,
1324
+ disconnected or replaced-session readings expose unknown occupancy, never zero.
1325
+ `context.gate` defaults to 0 (off); `context.stale_min` defaults to 30.
1326
+ A valid above-gate reading emits one metadata-only `context_pressure` event
1327
+ and one crossing notice. Invalid or stale readings do not rearm a crossing;
1328
+ a valid below-gate reading or a new native session does. A crossing held by pause or recovery remains unlatched and is reconsidered after release using only a still-fresh, current-session reading; no new native sample is required.
1329
+
1330
+ Context and quota checkpoints share the same request/wait path. Only one
1331
+ request per peer may wait at a time. Context responses must carry the supplied
1332
+ `request_id` and match the attached peer and native session. Timeout,
1333
+ disconnection, recovery hold, shutdown and session replacement invalidate the
1334
+ request. Context requests leave quota records, task assignments, pause state,
1335
+ native compaction settings and session ownership unchanged.
1336
+
1337
+ A valid non-private context response is saved in the state directory as a
1338
+ 0600 `context-checkpoint-<peer>.json` note. With memory enabled, its text is
1339
+ also saved as a handover note only after the active-turn, task PII and text
1340
+ pattern checks. It is never broadcast to peers or quoted in logs/events.
1341
+ Private turns and all non-approved PII tasks associated with the peer as owner or reviewer, including tasks already in review, block requests and completion. The current routing PII policy is rechecked at both boundaries. The operator chooses
1342
+ whether to continue the native session or restart; this change provides no
1343
+ automatic session replacement. Status, tail and dashboard expose readings with
1344
+ source, measurement time and freshness beside quota information.
1345
+
1346
+ Release 0.12.16 uses control protocol 14. Recovery sources 9 through 13 remain
1347
+ supported; protocol 13 identifies releases 0.12.4 through 0.12.15, while
1348
+ 0.12.16 uses protocol 14.
1349
+
1350
+ ## Operator console and conductor (issues #190, #191, #193, #194, #195)
1351
+
1352
+ `ahub console` owns one authenticated console connection. It shares tail's
1353
+ renderer, renders a DECSTBM stream and footer, and provides optional alternate
1354
+ screen panels. `up` opens it only with terminal input and output, unless
1355
+ `--no-console` is given. Tail remains a plain stream. Console commands use an
1356
+ allowlisted argv with stdin closed; lifecycle and native launches are excluded.
1357
+ Allow options require confirmation and never interpret nonempty input as an
1358
+ approval. Deny is direct. Pending requests include expiry and are withdrawn by
1359
+ `permission_closed` on any answer or cancellation. Answer audits have no title.
1360
+ Panels expose Peers, Approvals, Tasks, Queue and Events with bounded polling and
1361
+ Unicode cell widths, fall back below 80x24, and restore terminal state on exit.
1362
+
1363
+ Agent shell CLI calls connect as the detected peer in tools mode. Hub launches
1364
+ set `AGENTHUB_PEER_ID`; the pinned installed Codex shell injects
1365
+ `CODEX_THREAD_ID` after environment filtering, and Claude uses `CLAUDECODE`.
1366
+ Malformed or conflicting markers fail closed. Console-only commands are denied
1367
+ before connecting, with no as-user escape. Refusals enter a bounded ids-only
1368
+ local audit spool which a running daemon consumes; this preserves the no-connect
1369
+ rule while making refusals visible on the console and in the log. A stopped
1370
+ daemon cannot show a live notice; it consumes remaining records on startup.
1371
+ This does not establish a security boundary against an unrestricted shell
1372
+ which can read the token. Native source evidence and live probe results remain
1373
+ separate in the smoke ledger.
1374
+
1375
+ Exactly one explicit conductor role may be configured. Default-allow
1376
+ capabilities do not grant it; role authority is checked on each operation and
1377
+ refreshed on a subsequent connection. Shared MCP tools provide public status,
1378
+ public task history, actor-preserving assign/escalate, headless local/Kimi/Pi
1379
+ start, and conductor-owned persistent holds. TUI starts return launch commands.
1380
+ Assign/escalate need `assign` if the peer has a capabilities entry. Approval
1381
+ answers, durable queue resolution, budget overrides and lifecycle remain human
1382
+ operations. A conductor cannot release human or budget holds. Audit events are
1383
+ ids-only and report counts their actions.
1384
+
1385
+ The conductor feed defaults to own tasks, also supports all and off, and uses
1386
+ the existing bus digest window. Repeated queued task/kind milestones replace
1387
+ the earlier pending notice; accepted or in-flight deliveries are preserved.
1388
+ Structured milestones carry public titles or PII stubs, never raw history
1389
+ notes or check output. Only aged approval summaries and needs-review delivery
1390
+ ids are important. Role/feed revocation withdraws pending feed notices. A
1391
+ completed task set produces one round notice until a new task joins; subsequent
1392
+ rounds count only newly joined tasks. Pure status supervision queues wait for
1393
+ the existing digest deadline rather than flushing at the ordinary batch-count
1394
+ threshold. Mixed traffic retains the ordinary admission behavior. The existing
1395
+ 10-original digest ceiling and 200-entry queue ceiling still bound admission;
1396
+ larger windows may require multiple bounded deliveries.
1397
+
1398
+ Supervision cost counts completed native turns that received feed notices,
1399
+ with whole-turn token readings when available. Other work may share a turn;
1400
+ these readings are not per-notice token attribution. Missing measurements are
1401
+ unknown. Native TUI conductor, sandbox and feed-on smoke results must be recorded
1402
+ as observed outcomes, separately from unit/fake protocol tests.
1403
+
1404
+ Managed Claude launches with turn-free facts, task-idle sweeps or a conductor
1405
+ role install native observation hooks. A conductor receives them even when facts
1406
+ injection and task-idle sweeps are disabled; ordinary non-opt-in launches remain
1407
+ unchanged. SessionStart registers the native
1408
+ session and UserPromptSubmit starts observation; PreToolUse keeps a tool turn
1409
+ active and Stop closes it. Explicit caller settings remain authoritative and
1410
+ produce a warning when they replace these hooks. An ordinary terminal records a
1411
+ private launcher identity without creating an Orca terminal-recovery record.
1412
+ The facts control request accepts session/start/pre/post/stop phases and carries
1413
+ nativeInstanceId and nativeLaunchId alongside sessionId and transcriptPath.
1414
+ The command hook forwards launcher identity only for its matching state directory
1415
+ and peer; library calls targeting another hub do not inherit that identity.
1416
+ The daemon fences observations to its current instance and launcher, validates
1417
+ the session transcript, and rejects stale stop/post observations. Native prompt
1418
+ text is never included in those requests or events.
1419
+
1420
+ A genuine native Stop requires the current daemon/launcher binding and an actual
1421
+ assistant end_turn transcript message at or after the native turn's first start,
1422
+ distinct from the completed-message baseline recorded at that start. Later tool
1423
+ activity does not move this turn-start boundary. Missing start evidence or a
1424
+ start from another session, launch or channel claim leaves completion unknown.
1425
+ Accepted completion consumes that start before the peer becomes idle, so another
1426
+ message cannot reuse it. A valid bound Stop request receives one prompt
1427
+ acknowledgement with pending=true, within the existing two-second hook deadline.
1428
+ The acknowledgement records observation only, never completion. After the hook
1429
+ can return, a deferred observer waits up to 1200 monotonic milliseconds for its
1430
+ transcript append to become visible. Each read and final consumption revalidate
1431
+ the captured start, session,
1432
+ launch, peer and claim. A seen previous-turn baseline still waits while a new
1433
+ current start exists; a consumed duplicate remains a no-op. Timeout, malformed or
1434
+ oversized evidence and superseded context leave completion unknown. This wait
1435
+ does not retry a user action or relax authority.
1436
+ The observer sends no second request reply and never changes global hook settings.
1437
+ The live harness also binds its final receipt to the current private launch id.
1438
+ Its opaque
1439
+ deduplication id binds session, launch and message; transport replacement or new
1440
+ activity cannot turn a replay into another completion. Unbound or idless legacy
1441
+ Stop events cannot certify completion or finish supervision. A current private
1442
+ launcher marker takes precedence over an older terminal-recovery launch id for
1443
+ native observation, while preserving the recovery records and their authority.
1444
+ Claude turn reports
1445
+ prefer unique native Stop events; historical logical-state counts are labelled
1446
+ as such, and an unobserved completion remains unknown. Channel idle, watchdog
1447
+ expiry, board approval and a tool reply do not independently establish native
1448
+ completion. The live harness waits for the current fixture/instance/session's
1449
+ final transcript end_turn and turn_duration after its last review before
1450
+ reporting a completed case or terminating its native TUI. It also requires the
1451
+ matching opaque native completion event, an idle peer and settled delivery rows.
1452
+
1453
+ These additions use control protocol 15. Supported recovery sources include
1454
+ protocol 14 (0.12.16) alongside the previous source protocols.
1455
+
1456
+ For release 0.12.17 only, the user explicitly accepted deferral of real Kimi ACP
1457
+ new-tool invocation and ordinary-role refusal T0 evidence on 2026-10-09. The
1458
+ managed OAuth account rejected its prompt with HTTP 403 for the weekly quota
1459
+ before any tool event; reset time is unknown and no configured native alternative
1460
+ was found. This prerequisite remains unverified. Actual Claude/Codex conductor
1461
+ workflows, Claude shell/channel observations, shared MCP/fake-adapter authority
1462
+ checks, real Kimi ACP initialization/session creation and full source gates are
1463
+ separately qualified. The deferral authorizes no purchase, credentials,
1464
+ provider/configuration change, account retry or authority relaxation. Once quota
1465
+ is available under an authorized account, the same isolated read-only status-tool
1466
+ and ordinary-role refusal probe must record a native tool event and daemon result
1467
+ before this Kimi prerequisite can be marked verified.
@@ -0,0 +1,123 @@
1
+ # Agent shell identity T0, issues #193 and #194
2
+
3
+ ## Evidence collected
4
+
5
+ - Installed CLI reports `codex-cli 0.146.0` (`codex --version`, 2026-10-09). `codex app-server --help` supports configuration overrides and WebSocket listeners; its example names `shell_environment_policy.inherit=all`.
6
+ - The [version-pinned native environment implementation](https://github.com/openai/codex/blob/rust-v0.146.0/codex-rs/protocol/src/shell_environment.rs) applies inheritance, default exclusions, custom exclusions, explicit values, and `include_only` in that order. It then injects `CODEX_THREAD_ID`. `AGENTHUB_PEER_ID` is an ordinary inherited variable and can be removed by filtering. The native thread marker is injected after filtering, which supplies a fallback without widening the user's environment policy.
7
+ - The [version-pinned core wrapper](https://github.com/openai/codex/blob/rust-v0.146.0/codex-rs/core/src/exec_env.rs) explicitly documents thread-marker injection even with `include_only`.
8
+ - The [current configuration reference](https://learn.chatgpt.com/docs/config-file/config-reference) describes inheritance and filtering, and now describes a `filters` map alongside legacy `exclude` and `include_only`. Current documentation is not evidence that every new option exists in installed 0.146.0.
9
+ - The hub's Codex adapter passes a child environment even without optional launch values. It now sets its own marker and removes stale vendor markers inherited from the caller. Recovery authority remains scrubbed by `childEnv`.
10
+
11
+ ## Native Codex observations (2026-10-09)
12
+
13
+ - Real installed `codex exec --json --model gpt-5.5 --sandbox workspace-write`
14
+ completed a single fixed Python shell probe in an isolated git fixture.
15
+ With explicit `inherit=all`, the hub marker, native thread marker and harmless
16
+ canary were present; a Claude marker was absent. The disposable loopback
17
+ responder was reachable by the harness, while the native shell request failed
18
+ with `URLError`, errno 1. No sandbox or network exception was requested.
19
+ - A second real run used `inherit=none`, an explicit PATH and
20
+ `include_only=["PATH"]`. The hub marker, native marker and canary still appeared
21
+ in the observed shell. This does not establish that filtering removed them;
22
+ the difference from the pinned environment construction is unresolved in this
23
+ installed host environment. The fallback is supported by source and marker
24
+ presence, without widening the caller's configured policy.
25
+ - A real native MCP call under `workspace-write` completed `agent-hub.hub_status`
26
+ against the candidate bundle and an isolated conductor daemon. Native JSONL
27
+ recorded the MCP call completing, and the daemon independently recorded a
28
+ codex `conduct/status` event. Only this exact hub tool was approved through the
29
+ existing per-tool `approval_mode=approve` configuration; its daemon role check
30
+ still applied. This proves MCP access separately from the blocked shell path.
31
+ - The first run with the user's configured `gpt-6.1-sol` failed before tools:
32
+ that slug was rejected by the installed ChatGPT-account endpoint. The probe's
33
+ explicit model selection did not change the user's configuration.
34
+
35
+ ## Native Claude observations (2026-10-09)
36
+
37
+ - The final observer proof used source `20ad5e6`. Its read-only feed-off
38
+ continuation verified one completed native turn, 263,413 cached-inclusive
39
+ tokens, a matching opaque daemon Stop receipt, native idle and no pending
40
+ deliveries. The fresh feed-own run verified two actual owner completions and
41
+ independent file reviews, nine native finals matched by nine daemon Stops,
42
+ 27 unique usage records and 1,888,242 tokens. Eight completed turns contained
43
+ supervision, totalling 1,292,922 tokens. The receipt matched the final native
44
+ message, session, launcher and instance; earlier incomplete captures below
45
+ were not retroactively marked complete.
46
+
47
+ - In the original captures, Claude Code 2.1.295, Opus 5.5, ran the candidate MCP tools in real PTYs
48
+ after its five-hour quota reset. Both feed-off and feed-own fixtures started
49
+ local and headless Pi, proposed two tasks, reassigned Beta to Pi, held and
50
+ released both peers, and independently read the exact ALPHA/BETA files before
51
+ approving their tasks. The console independently audited four allow-once
52
+ answers in each fixture, covering the two writes and bounded file checks.
53
+ - The real feed-off TUI displayed inbound review requests; the feed-own TUI
54
+ displayed actual supervision and review pushes. MCP success and channel
55
+ delivery were observed separately. A transient startup warning about the
56
+ server name did not prevent the later observed channel deliveries.
57
+ - In the feed-own native TUI, the operator submitted the fixed Python probe
58
+ through `!`. Its actual command output was `AGENTHUB_PEER_ID=true`,
59
+ `CLAUDECODE=true`, `CLAUDE_CODE_SESSION_ID=true`, `CODEX_THREAD_ID=false`.
60
+ This is shell output captured before Claude's interpretation of it.
61
+ - A subsequent feed-off run pinned to `da2ebbc` executed the same fixed probe
62
+ through the model's actual native Bash tool and separately through `!`.
63
+ Both actual command outputs showed the same three Claude/hub markers present
64
+ and `CODEX_THREAD_ID` absent. This closes the separate model-shell observation;
65
+ neither command printed environment values.
66
+ - The original harness accepted hub idle too early and stopped both last
67
+ review turns before a native final answer. Native transcripts independently
68
+ show two completed turns in feed-off and five in feed-own, with the last
69
+ assistant message still `tool_use` in each case. The board approvals are real;
70
+ complete final-turn verification and production supervision accounting were
71
+ incomplete for those captures. The harness now requires the current daemon instance, its session,
72
+ the fixture-specific transcript, a final `end_turn` and `turn_duration` after
73
+ the last actual review. Missing measurements remain unknown.
74
+ - The `da2ebbc` run verified the final native message UUID, message id,
75
+ `end_turn`, subsequent `turn_duration`, daemon instance, session and launcher.
76
+ Its complete transcript contains five completed turns and 21 unique usage
77
+ records, totalling 1,451,782 tokens including cached input. The daemon certified
78
+ only one native Stop, however. Its log refused the other completions as an
79
+ unavailable completed transcript message or completion predating the current
80
+ turn; the final refusal occurred before cleanup. A normal review receipt
81
+ remained queued. Actual task/file/final-answer evidence is valid, while full
82
+ runtime accounting and delivery settlement remain incomplete.
83
+ - The `6034dd9` read-only continuation preserved both approved tasks and exact
84
+ files. It completed one native turn with four unique usage records and 261,483
85
+ cached-inclusive tokens, but the matching daemon Stop remained absent. After
86
+ a bounded retry, the Stop handler logged unavailable completion at
87
+ 01:39:04.087 UTC; the native stop summary/duration appeared at 01:39:04.095 UTC.
88
+ With no overlapping operator prompt, this strongly supports a transcript
89
+ visibility barrier while the native hook waits. The strict harness timed out
90
+ incomplete and stopped its owned processes. The final message's thinking/text
91
+ rows carried identical usage counters in these observed transcripts; this
92
+ observation does not replace the final text/duration requirement.
93
+
94
+ ## Plain Claude and Kimi observations (2026-10-09)
95
+
96
+ - Plain Claude 2.1.295, launched with the candidate MCP configuration but no
97
+ development-channel flag, completed one native `hub_status` call with an
98
+ independent daemon Claude status audit. Its only native tool calls were
99
+ ToolSearch and that status call. A unique operator push was bridge-accepted,
100
+ but no native pushed user row or response appeared in the observed
101
+ 20.075-second window. This bounded negative observation is not a universal
102
+ claim about channel support. Actual channel-enabled review/supervision pushes
103
+ were observed separately in the conductor cases. The plain probe and cleanup
104
+ took 56.724 seconds and made no model file, shell, task or settings changes.
105
+ - Real Kimi Code CLI 2.1.1 answered ACP `initialize` and created native session
106
+ `session_17eeb9dc-38d3-4685-b1aa-3db6a89b465c`. Its single prompt failed with
107
+ protocol error `-32000`, HTTP 403, for the managed account's weekly usage
108
+ limit before any requested tool event. No task-list/status result or ordinary
109
+ peer's conductor refusal was obtained. The owning CLI's safe provider list
110
+ showed only `managed:kimi-code`, type `kimi`, four models, OAuth, default
111
+ `kimi-code/k3`. No distinct configured provider was found. The reset time
112
+ remains unknown; no purchase, credential/configuration change or model retry
113
+ was attempted. Owned native processes and the fixture daemon were stopped.
114
+
115
+ ## Release disposition and remaining bounded live probe
116
+
117
+ - On 2026-10-09 the user explicitly approved release 0.12.17 with only the Kimi native new-tool and ordinary-role refusal T0 evidence deferred. Kimi coverage remains unverified. This decision defers no other source/native gate and authorizes no purchase, credentials, provider/configuration change, account retry or authority relaxation.
118
+ - Kimi ACP MCP: after native provider access is available under an authorized account, request the same read-only task list/status and ordinary-role refusal probes in isolated sessions. Verify actual native tool events/results and the daemon reply before marking this prerequisite verified. The existing quota failure proves neither new-tool access nor refusal behavior.
119
+
120
+ ## Verification limits
121
+
122
+ - Codex shell and MCP probes used the root agent's serial resource slot. Claude `!`, model shell, enabled-channel pushes and the bounded plain-session comparison were observed in subsequent serial native cases. Kimi new-tool access remains unverified because its real account rejected the prompt before tools.
123
+ - New unit and fake-adapter tests are prepared but have not been run by this worker. The root agent owns the final immutable-head checks.
@@ -0,0 +1,168 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "documents": {
4
+ "README.md": {
5
+ "verifiedAgainst": "c3d1e22f9b200cbe3f484c9e73d30bad046c9221",
6
+ "paths": [
7
+ "package.json",
8
+ "src/",
9
+ "templates/",
10
+ "plugins/agent-hub/.claude-plugin/plugin.json"
11
+ ]
12
+ },
13
+ "docs/security.md": {
14
+ "verifiedAgainst": "c3d1e22f9b200cbe3f484c9e73d30bad046c9221",
15
+ "paths": [
16
+ "src/",
17
+ "templates/"
18
+ ]
19
+ },
20
+ "docs/operations.md": {
21
+ "verifiedAgainst": "c3d1e22f9b200cbe3f484c9e73d30bad046c9221",
22
+ "paths": [
23
+ "package.json",
24
+ "src/",
25
+ "templates/",
26
+ "scripts/overlaps.ts",
27
+ "scripts/check-docs.mjs",
28
+ "scripts/seeded-check.ts",
29
+ "scripts/seeds.json",
30
+ ".github/workflows/check.yml",
31
+ "scripts/ci-gates.mjs",
32
+ "scripts/ci-reuse.mjs",
33
+ ".github/workflows/release.yml",
34
+ "scripts/smoke-conductor.ts"
35
+ ]
36
+ },
37
+ "docs/quickstart.md": {
38
+ "verifiedAgainst": "c3d1e22f9b200cbe3f484c9e73d30bad046c9221",
39
+ "paths": [
40
+ "package.json",
41
+ "src/",
42
+ "templates/"
43
+ ]
44
+ },
45
+ "docs/agent-notes/adapters.md": {
46
+ "verifiedAgainst": "20ad5e68fd1264fa43dd7426f60cb0c8211d6555",
47
+ "paths": [
48
+ "src/adapters/",
49
+ "src/pi/",
50
+ "src/hub/peers.ts",
51
+ "src/hub/child-process.ts",
52
+ "src/hub/lifecycle.ts",
53
+ "src/memory/capture.ts",
54
+ "scripts/benchmarks/teardown.ts",
55
+ "src/hub/daemon.ts"
56
+ ]
57
+ },
58
+ "docs/agent-notes/benchmarks.md": {
59
+ "verifiedAgainst": "8b7d7254f1bb5602ec5d658e7814e236f8fb43e7",
60
+ "paths": [
61
+ "scripts/benchmarks/",
62
+ "test/benchmarks/",
63
+ "src/pi/",
64
+ "src/adapters/acp.ts",
65
+ "src/adapters/pi.ts",
66
+ "src/models/relay.ts",
67
+ "src/local/"
68
+ ]
69
+ },
70
+ "docs/agent-notes/budget.md": {
71
+ "verifiedAgainst": "20ad5e68fd1264fa43dd7426f60cb0c8211d6555",
72
+ "paths": [
73
+ "src/hub/budget.ts",
74
+ "src/cli/statusline-tee.ts",
75
+ "src/hub/bus.ts",
76
+ "src/hub/delivery-journal.ts",
77
+ "src/hub/restart.ts",
78
+ "src/hub/context-window.ts",
79
+ "src/hub/daemon.ts"
80
+ ]
81
+ },
82
+ "docs/agent-notes/bus.md": {
83
+ "verifiedAgainst": "20ad5e68fd1264fa43dd7426f60cb0c8211d6555",
84
+ "paths": [
85
+ "src/hub/bus.ts",
86
+ "src/hub/envelope.ts",
87
+ "src/hub/limits.ts",
88
+ "src/hub/delivery-journal.ts",
89
+ "src/hub/inference.ts",
90
+ "src/hub/daemon.ts",
91
+ "src/hub/supervision.ts"
92
+ ]
93
+ },
94
+ "docs/agent-notes/daemon.md": {
95
+ "verifiedAgainst": "20ad5e68fd1264fa43dd7426f60cb0c8211d6555",
96
+ "paths": [
97
+ "src/hub/daemon.ts",
98
+ "src/hub/control-client.ts",
99
+ "src/adapters/claude-channel.ts",
100
+ "src/hub/restart.ts",
101
+ "src/hub/recovery-store.ts",
102
+ "src/hub/snapshots.ts",
103
+ "src/hub/manager.ts",
104
+ "src/hub/lifecycle.ts",
105
+ "src/hub/crash.ts",
106
+ "src/cli/setup.ts",
107
+ "src/cli/upgrade*.ts",
108
+ "src/cli/terminal-recovery.ts",
109
+ "src/cli/recovery-package.ts",
110
+ "src/cli/facts-hook.ts",
111
+ "src/pi/process-signature.ts",
112
+ "src/hub/context-window.ts",
113
+ "src/hub/conductor.ts",
114
+ "src/cli/console.ts",
115
+ "src/cli/console-state.ts",
116
+ "src/cli/identity.ts",
117
+ "src/cli/identity-audit.ts"
118
+ ]
119
+ },
120
+ "docs/agent-notes/local-worker.md": {
121
+ "verifiedAgainst": "8b7d7254f1bb5602ec5d658e7814e236f8fb43e7",
122
+ "paths": [
123
+ "src/adapters/local-worker.ts",
124
+ "src/local/",
125
+ "src/memory/capture.ts",
126
+ "src/hub/facts.ts"
127
+ ]
128
+ },
129
+ "docs/agent-notes/models.md": {
130
+ "verifiedAgainst": "8b7d7254f1bb5602ec5d658e7814e236f8fb43e7",
131
+ "paths": [
132
+ "src/models/",
133
+ "src/hub/inference.ts",
134
+ "src/omniroute/",
135
+ "src/switchyard/"
136
+ ]
137
+ },
138
+ "docs/agent-notes/tasks.md": {
139
+ "verifiedAgainst": "c3d1e22f9b200cbe3f484c9e73d30bad046c9221",
140
+ "paths": [
141
+ "src/hub/tasks.ts",
142
+ "src/hub/board.ts",
143
+ "src/hub/routing.ts",
144
+ "src/hub/hub-tools.ts",
145
+ "src/hub/task-sweep.ts",
146
+ "src/cli/launch.ts",
147
+ "src/cli/main.ts",
148
+ "src/cli/preview.ts",
149
+ "src/hub/conductor.ts",
150
+ "src/hub/supervision.ts"
151
+ ]
152
+ },
153
+ "docs/agent-notes/tests.md": {
154
+ "verifiedAgainst": "2a274289e280252b264716da3a1112f67b0c278e",
155
+ "paths": [
156
+ "test/",
157
+ "scripts/check.sh",
158
+ "scripts/hang-watch.sh",
159
+ "scripts/seeded-check.ts",
160
+ "scripts/seeds.json",
161
+ "scripts/ci-gates.mjs",
162
+ "scripts/ci-reuse.mjs",
163
+ ".github/workflows/check.yml",
164
+ ".github/workflows/release.yml"
165
+ ]
166
+ }
167
+ }
168
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@staix/agent-hub",
3
- "version": "0.12.15",
3
+ "version": "0.12.17",
4
4
  "description": "Native multi-agent hub: Claude Code, Codex, Kimi Code, Pi and local inference as peers in one project",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "agent-hub",
3
- "version": "0.12.15",
3
+ "version": "0.12.17",
4
4
  "description": "Channel between Claude Code and the agent-hub daemon: peer messages from Codex, Kimi and the local worker arrive as channel events; hub_send replies.",
5
5
  "author": {
6
6
  "name": "Young Joon Lee",