omnius 1.0.697 → 1.0.698

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -292930,6 +292930,9 @@ var DECAY_TAU2 = {
292930
292930
  var DEFAULT_INLINE_KG_MAX_BYTES = 512 * 1024 * 1024;
292931
292931
  var DEFAULT_INLINE_EPISODES_MAX_BYTES = 1024 * 1024 * 1024;
292932
292932
 
292933
+ // packages/execution/dist/tools/diagnostic.js
292934
+ init_process_async();
292935
+
292933
292936
  // packages/execution/dist/tools/background-task.js
292934
292937
  init_process_kill();
292935
292938
  init_process_lifecycle();
@@ -32878,6 +32878,46 @@
32878
32878
  }
32879
32879
  ]
32880
32880
  },
32881
+ {
32882
+ "id": "guide.work-orders-runtime-health-remediation-live-telegram-log-review-2026-09-05-uppercase",
32883
+ "kind": "guide",
32884
+ "title": "Telegram field review after publication — September 5, 2026",
32885
+ "summary": "Scope: read-only review requested after the user observed improvement. Runtime package is 1.0.697. The launcher started at 01:05 PDT; its current runtime child and Telegram poll owner is PID 4160370, started at 13:33 PDT. Two processes are the launcher/child relationship, not evidence of competing pollers.",
32886
+ "keywords": [
32887
+ "work",
32888
+ "orders",
32889
+ "runtime",
32890
+ "health",
32891
+ "remediation",
32892
+ "LIVE",
32893
+ "TELEGRAM",
32894
+ "LOG",
32895
+ "REVIEW",
32896
+ "2026",
32897
+ "09",
32898
+ "05",
32899
+ "md"
32900
+ ],
32901
+ "maturity": "internal",
32902
+ "audiences": [
32903
+ "maintainer",
32904
+ "large-context-agent"
32905
+ ],
32906
+ "layer": "documentation",
32907
+ "interfaces": [
32908
+ {
32909
+ "type": "file",
32910
+ "target": "docs/work-orders/runtime-health-remediation/LIVE-TELEGRAM-LOG-REVIEW-2026-09-05.md"
32911
+ }
32912
+ ],
32913
+ "references": [
32914
+ {
32915
+ "type": "documentation",
32916
+ "target": "docs/work-orders/runtime-health-remediation/LIVE-TELEGRAM-LOG-REVIEW-2026-09-05.md",
32917
+ "relation": "canonical-artifact"
32918
+ }
32919
+ ]
32920
+ },
32881
32921
  {
32882
32922
  "id": "guide.work-orders-runtime-health-remediation-readme-uppercase",
32883
32923
  "kind": "guide",
@@ -35015,6 +35055,120 @@
35015
35055
  }
35016
35056
  ]
35017
35057
  },
35058
+ {
35059
+ "id": "guide.work-orders-runtime-health-remediation-wo-46-shell-result-authority-uppercase",
35060
+ "kind": "guide",
35061
+ "title": "WO-46: Preserve process authority for shell observations",
35062
+ "summary": "Status: repository repair complete and delivered to origin/main. Publication and live validation remain with the user.",
35063
+ "keywords": [
35064
+ "work",
35065
+ "orders",
35066
+ "runtime",
35067
+ "health",
35068
+ "remediation",
35069
+ "WO",
35070
+ "46",
35071
+ "shell",
35072
+ "result",
35073
+ "authority",
35074
+ "md"
35075
+ ],
35076
+ "maturity": "internal",
35077
+ "audiences": [
35078
+ "maintainer",
35079
+ "large-context-agent"
35080
+ ],
35081
+ "layer": "documentation",
35082
+ "interfaces": [
35083
+ {
35084
+ "type": "file",
35085
+ "target": "docs/work-orders/runtime-health-remediation/WO-46-shell-result-authority.md"
35086
+ }
35087
+ ],
35088
+ "references": [
35089
+ {
35090
+ "type": "documentation",
35091
+ "target": "docs/work-orders/runtime-health-remediation/WO-46-shell-result-authority.md",
35092
+ "relation": "canonical-artifact"
35093
+ }
35094
+ ]
35095
+ },
35096
+ {
35097
+ "id": "guide.work-orders-runtime-health-remediation-wo-47-diagnostic-execution-contract-uppercase",
35098
+ "kind": "guide",
35099
+ "title": "WO-47: Validate and execute diagnostics without blocking the agent",
35100
+ "summary": "Status: repository repair complete and delivered to origin/main. Publication and live validation remain with the user.",
35101
+ "keywords": [
35102
+ "work",
35103
+ "orders",
35104
+ "runtime",
35105
+ "health",
35106
+ "remediation",
35107
+ "WO",
35108
+ "47",
35109
+ "diagnostic",
35110
+ "execution",
35111
+ "contract",
35112
+ "md"
35113
+ ],
35114
+ "maturity": "internal",
35115
+ "audiences": [
35116
+ "maintainer",
35117
+ "large-context-agent"
35118
+ ],
35119
+ "layer": "documentation",
35120
+ "interfaces": [
35121
+ {
35122
+ "type": "file",
35123
+ "target": "docs/work-orders/runtime-health-remediation/WO-47-diagnostic-execution-contract.md"
35124
+ }
35125
+ ],
35126
+ "references": [
35127
+ {
35128
+ "type": "documentation",
35129
+ "target": "docs/work-orders/runtime-health-remediation/WO-47-diagnostic-execution-contract.md",
35130
+ "relation": "canonical-artifact"
35131
+ }
35132
+ ]
35133
+ },
35134
+ {
35135
+ "id": "guide.work-orders-runtime-health-remediation-wo-48-conversation-result-evidence-uppercase",
35136
+ "kind": "guide",
35137
+ "title": "WO-48: Retain command evidence in conversation results",
35138
+ "summary": "Status: repository repair complete and delivered to origin/main. Publication and live validation remain with the user.",
35139
+ "keywords": [
35140
+ "work",
35141
+ "orders",
35142
+ "runtime",
35143
+ "health",
35144
+ "remediation",
35145
+ "WO",
35146
+ "48",
35147
+ "conversation",
35148
+ "result",
35149
+ "evidence",
35150
+ "md"
35151
+ ],
35152
+ "maturity": "internal",
35153
+ "audiences": [
35154
+ "maintainer",
35155
+ "large-context-agent"
35156
+ ],
35157
+ "layer": "documentation",
35158
+ "interfaces": [
35159
+ {
35160
+ "type": "file",
35161
+ "target": "docs/work-orders/runtime-health-remediation/WO-48-conversation-result-evidence.md"
35162
+ }
35163
+ ],
35164
+ "references": [
35165
+ {
35166
+ "type": "documentation",
35167
+ "target": "docs/work-orders/runtime-health-remediation/WO-48-conversation-result-evidence.md",
35168
+ "relation": "canonical-artifact"
35169
+ }
35170
+ ]
35171
+ },
35018
35172
  {
35019
35173
  "id": "guide.work-orders-telegram-dmn-wo-22-dmn-outreach-and-learning-uppercase",
35020
35174
  "kind": "guide",
package/docs/DISCOVERY.md CHANGED
@@ -534,6 +534,7 @@ Daemon equivalents are `GET /v1/discovery/bootstrap`, `GET /v1/discovery?q=<inte
534
534
  | `guide.work-orders-runtime-health-remediation-h1-h2-h7-telegram-continuity-evidence-uppercase` | H1, H2, and H7 Telegram continuity evidence | Scope: deterministic Telegram stream ownership, immutable task-local routing, exact-once text delivery evidence, bounded long-poll client replacement, bridge restart authority, and task-local delegated-agent runtime ownership. |
535
535
  | `guide.work-orders-runtime-health-remediation-h6-main-tui-wo-05-confirmation-evidence-uppercase` | H6 Main-TUI and WO-05 Confirmation Evidence | This closure wires the top-level TUI FullSubAgentTool to live parent-context headroom and the exact owning task cancellation signal. It also confirms the WO-05 completion authority after the interruption, safe-media, workboard, and canonical-context changes. |
536
536
  | `guide.work-orders-runtime-health-remediation-live-telegram-log-review-2026-09-03-uppercase` | Live Telegram Log Review, 2026-09-03 | This review is read-only. It examines the persisted Telegram conversation and intake lifecycle records in /home/roko/Documents/Projects/Adjacent/telegramtest/.omnius. The running global Omnius package reports version 1.0.687. This means the observations describe the deployed pre-publication runtime, not the newer source commits on this repository's main bran |
537
+ | `guide.work-orders-runtime-health-remediation-live-telegram-log-review-2026-09-05-uppercase` | Telegram field review after publication — September 5, 2026 | Scope: read-only review requested after the user observed improvement. Runtime package is 1.0.697. The launcher started at 01:05 PDT; its current runtime child and Telegram poll owner is PID 4160370, started at 13:33 PDT. Two processes are the launcher/child relationship, not evidence of competing pollers. |
537
538
  | `guide.work-orders-runtime-health-remediation-readme-uppercase` | Runtime Health Remediation Program | Program ID: RHR-2026-09-02 Status: active, implementation authorized Owner: Omnius runtime, orchestration, execution, memory, and Telegram packages External dependency: /home/roko/Desktop/ollama-unify Last reconciled: 2026-09-03 |
538
539
  | `guide.work-orders-runtime-health-remediation-telegram-longhaul-2026-09-04-uppercase` | September 4 Telegram long-haul remediation | Status: complete; all five root repairs verified and delivered to origin/main Baseline: Omnius main 69c4a160; installed package 1.0.694 Incident run: telegram-64ac9937dc7647a8-1788572475259-1 Observation window: 2026-09-04 18:41–20:43 PDT Owner: repository repair; user owns publication and subsequent live testing |
539
540
  | `guide.work-orders-runtime-health-remediation-tool-quality-2026-09-05-uppercase` | September 5 tool-quality and live-behavior follow-up | Status: WO-30 through WO-42 complete in repository; full verification passed; delivered to origin/main for user publication Repository baseline: 501e8394 on origin/main Verified source head: 7005aa41; subsequent closure changes are documentation only Observed running package: 1.0.695, verified from its actual executable/package path User scope: monitor live |
@@ -590,6 +591,9 @@ Daemon equivalents are `GET /v1/discovery/bootstrap`, `GET /v1/discovery?q=<inte
590
591
  | `guide.work-orders-runtime-health-remediation-wo-43-telegram-router-progress-boundary-uppercase` | WO-43: Keep router failures out of Telegram progress and repair routing contracts | Status: complete in repository; verified for user publication Priority: P1 Reported: September 5, 2026, after publication of 1.0.696 Source baseline: d8e6e2ed on origin/main |
591
592
  | `guide.work-orders-runtime-health-remediation-wo-44-evidence-backed-terminal-results-uppercase` | WO-44: Deliver an evidence-backed final result in Telegram | Status: complete in repository; publication and live acceptance remain with the user Priority: P1 Source baseline: 44fa69cc Observed runtime: 1.0.696, sibling telegramtest; read-only inspection |
592
593
  | `guide.work-orders-runtime-health-remediation-wo-45-native-ollama-tool-contract-uppercase` | WO-45: Preserve tools across the native Ollama transport | Status: complete — implementation, mocked transport verification and scoped Git delivery Priority: P1 for callers selecting native Ollama chat with tools Source baseline: af6fb8eb |
594
+ | `guide.work-orders-runtime-health-remediation-wo-46-shell-result-authority-uppercase` | WO-46: Preserve process authority for shell observations | Status: repository repair complete and delivered to origin/main. Publication and live validation remain with the user. |
595
+ | `guide.work-orders-runtime-health-remediation-wo-47-diagnostic-execution-contract-uppercase` | WO-47: Validate and execute diagnostics without blocking the agent | Status: repository repair complete and delivered to origin/main. Publication and live validation remain with the user. |
596
+ | `guide.work-orders-runtime-health-remediation-wo-48-conversation-result-evidence-uppercase` | WO-48: Retain command evidence in conversation results | Status: repository repair complete and delivered to origin/main. Publication and live validation remain with the user. |
593
597
  | `guide.work-orders-telegram-dmn-wo-22-dmn-outreach-and-learning-uppercase` | WO-22: DMN outreach, DM sharing, and outcome learning | On 2026-09-03 at 17:07 PDT the bot posted in the OMNIUS group without being addressed. The operator asked whether this was self-induced reflection. |
594
598
  | `guide.work-orders-telegram-dropbear-context-rca-workorder` | Telegram Dropbear Context Engineering RCA Work Order | Observed run: /home/roko/Documents/Projects/Adjacent/telegramtest/.omnius, run id 1782873796963-i5r7mv. |
595
599
  | `guide.work-orders-wo-am-gaps-uppercase` | Associative Memory Gap Work Orders | Generated: 2026-04-13 Source: Deep audit of multimodal associative memory systems Status: READY FOR IMPLEMENTATION |
@@ -0,0 +1,58 @@
1
+ # Telegram field review after publication — September 5, 2026
2
+
3
+ Scope: read-only review requested after the user observed improvement. Runtime package is **1.0.697**. The launcher started at 01:05 PDT; its current runtime child and Telegram poll owner is PID 4160370, started at 13:33 PDT. Two processes are the launcher/child relationship, not evidence of competing pollers.
4
+
5
+ Sources are under the sibling `telegram_test/.omnius/`. Selected records were copied and hashed before analysis; capture directories are `/tmp/omnius-latest-completion-audit-oeh__cit`, `/tmp/omnius-telegram-delivery-audit-20260905-2045`, and `/tmp/omnius-field-context-w8e603c3`. Timestamps below are PDT (UTC−07:00). This review does not send inference or Telegram requests or change the running workspace.
6
+
7
+ ## Confirmed improvement
8
+
9
+ The assessment run `telegram-64ac9937dc7647a8-1788640550740-1` executed from 13:35:50 to 13:41:07: 17 turns, 25 tool calls, about 316 seconds. Its task was to identify remaining work and validation for boutique-agent-services. Its completed status applies to that assessment; the project itself remains unfinished.
10
+
11
+ At **13:41:08.452**, Telegram received a concrete 1,494-character final answer with six remaining integration gaps and ordered next actions. It did not contain the previous opaque completion fallback or internal router rationale. The reported checks have supporting tool output: TypeScript exited zero and Vitest reported **21/21 passing**, exit zero. No project mutations were recorded for the assessment.
12
+
13
+ Evidence:
14
+
15
+ - `telegram-conversations/64ac9937dc7647a833ca.events.jsonl:1510`: final message 2767.
16
+ - `telegram-intake/98fe41d25e55150bca68.jsonl:836`: assessment consumed because the reply was delivered.
17
+ - `terminal-trajectories/telegram-64ac9937dc7647a8-1788640550740-1.json:8`: run outcome and scope.
18
+ - Captured debug records `2026-09-05T20-39-34-017Z-shell-d1e40f17f3.json:21` and `2026-09-05T20-39-35-953Z-shell-686a982eff.json:21`: actual check output.
19
+
20
+ ## Liveness, routing and context
21
+
22
+ Both recent requests moved from deferred to dispatched in about 352–353 ms. Routing/admission still took approximately 37–38 seconds before each execution began. The assessment used one strict routing retry and subsequently delivered its answer. The selected records do not identify the first retry's cause.
23
+
24
+ Twelve draft updates were accepted across the assessment and its continuation by the sampled delivery window, with no recorded rejection, rate limit or ambiguity. A draft-only delivery ledger remaining pending is expected; terminal state requires a final, fallback or receipt. The poll health snapshot had zero consecutive poll failures and one pending update corresponding to active work.
25
+
26
+ Continuous typing cannot be established from these records: they do not log every typing pulse. Current stdout/stderr point to `/dev/pts/3`; the ordinary `.omnius/logs` files are stale and must not be treated as current transport evidence.
27
+
28
+ The continuation `telegram-64ac9937dc7647a8-1788640948905-1` began at **13:42:28**. By the final **13:48:02** snapshot it retained the boutique-agent-services wiring objective, acknowledged 14 file reads and updated four relevant todos, marking discovery done and implementation in progress at 13:46:46. No source mutation, completed verification, ambiguous effect or terminal result was observed yet. Five completed main calls took 23–82 seconds each; the current call had been pending 59 seconds at capture, which does not establish a stall.
29
+
30
+ All **22 sampled main requests** were admitted: roughly 9,752–33,023 input tokens against 262,144 capacity, with at least 79.3% free capacity. The latest prompt dropped from 31,729 to 16,784 input tokens through canonical supersession of 20 messages, preserving current task continuity. No memory-compilation stage or capacity-driven compaction was exercised. Latest admission and todo evidence are in `context-window-dumps/2026-09-05T20-47-03-254Z-main-d47c4be5a1.json:87` and `:1265`.
31
+
32
+ The native `feature_workflow` was **not selected**: no workflow store, workflow tool call or protected workflow-owner context appeared in the captured records. This review establishes ordinary task continuity under the observed load, not live acceptance of native long-horizon workflow recovery or compaction.
33
+
34
+ ## Remaining findings
35
+
36
+ ### F1 — P2: source reads are falsely classified as runtime failures
37
+
38
+ Two successful `cat` commands returned native transcript `status: success`, `exit_code: 0`, no stderr and no mutations, but the runner changed the result to `Runtime exception: throw new Error`. The phrase was source code being read.
39
+
40
+ Root: `packages/orchestrator/src/agenticRunner.ts:32094` applies semantic stdout failure detection to successful shell output; the detector at line 15911 treats a literal `throw new Error` as an exception without a command-role or typed-process-receipt boundary. The contradiction survives in `context-window-dumps/2026-09-05T20-37-42-904Z-main-06ca92faa6.json:915` and `:942`. Raw typed receipts were not retained for this conversation run, so the field conclusion relies on its native transcript and non-mutation records.
41
+
42
+ Required repair: preserve successful observation commands using host-owned command/receipt classification. Keep actual nonzero exits, timeouts and declared verification failures authoritative. Regressions should execute source-reading commands whose output contains exception/build/test-failure literals and confirm they remain observations, alongside real failing-command controls.
43
+
44
+ ### F2 — P2: malformed diagnostic arguments escape validation and crash
45
+
46
+ The model supplied `steps` as a JSON-encoded string. `DiagnosticTool` casts it to `string[]` at `packages/execution/src/tools/diagnostic.ts:53`, then calls `.filter` at line 60. It has no executable input schema or value-level validator, so the runner's required-field-only fallback admits the value. The recorded error is `requestedSteps.filter is not a function` in captured debug record `2026-09-05T20-38-27-750Z-diagnostic-537aa683ec.json:22`.
47
+
48
+ Required repair: introduce an executable input contract and validate direct invocation too; return an actionable argument error before executing anything. Test string/null/object steps, invalid members, valid subsets and explicit unavailable checks. This run recovered by executing TypeScript and Vitest directly.
49
+
50
+ ### F3 — coverage limitation: conversation checks do not populate the terminal evidence ledger
51
+
52
+ The new bound terminal report exists in `completion-finalizations/telegram-64ac9937dc7647a8-1788640550740-1.json:83`, but its structured changes/checks/observations are empty and it explicitly reports the absent evidence snapshot. Telegram's chat profile selects conversation interaction mode; `writesUserTaskArtifacts()` at `agenticRunner.ts:6719` and startup at line 24954 leave that mode's completion ledger null.
53
+
54
+ This is a reporting-coverage limitation, not evidence of a failed persistence write or missing delivered answer in this run. The accepted model final was delivered. The action continuation has an open ledger with actual command evidence. A follow-up design should retain observational command receipts for conversation reports without granting task/mutation authority or manufacturing verification claims.
55
+
56
+ ## Assessment
57
+
58
+ No hard blocker was established in the sampled records. Useful final delivery, prompt continuity and queue dispatch improved; two tool defects caused avoidable failures but recovered. The implementation continuation remains active and must not be reported as completed. F1–F3 are findings from this review, not repaired or live-validated claims. Native workflow behavior and continuous typing still require the corresponding field evidence.
@@ -4,6 +4,19 @@
4
4
  **Checked-item rule:** code, focused tests, and named evidence must all exist
5
5
  **Last reconciled:** 2026-09-05
6
6
 
7
+ ## September 5 post-publication field review
8
+
9
+ [Runtime 1.0.697 field review](LIVE-TELEGRAM-LOG-REVIEW-2026-09-05.md): a useful assessment final was delivered at 13:41 PDT, supported by TypeScript and 21 passing tests. At the reviewed checkpoint, the next implementation run was active. No hard blocker was established; source-read false failures, malformed diagnostic input handling and conversation-mode report coverage triggered the repairs below. Native feature workflow selection and continuous typing are not established by this sample.
10
+
11
+ The user authorized end-to-end repair of the remaining defects; repository work is now complete:
12
+
13
+ - [x] [WO-46: shell result authority](WO-46-shell-result-authority.md).
14
+ - [x] [WO-47: diagnostic execution contract](WO-47-diagnostic-execution-contract.md).
15
+ - [x] [WO-48: conversation result evidence](WO-48-conversation-result-evidence.md).
16
+
17
+ Repairs `a3c26eca` (shell authority), `d4c37489` (validated asynchronous diagnostics), and `672bef53` (durable conversation evidence and terminal reporting) are delivered to `origin/main`. Clean workspace build passed. Execution: 1,820 passed / 3 existing skips; CLI: 2,575 passed. Orchestrator: 2,847 passed / 1 existing skip in the full run, with its sole outdated receipt fixture corrected and all 10 tests in that suite passing on rerun. All 7,243 distinct non-skipped affected-package tests have a passing result on the final implementation; the workorders retain exact commands, initial failures, limits and evidence paths. Publication and live acceptance remain with the user.
18
+
19
+
7
20
  ## September 5 terminal result and long-horizon follow-up
8
21
 
9
22
  Current terminal repair authority: [WO-44](WO-44-evidence-backed-terminal-results.md).
@@ -0,0 +1,42 @@
1
+ # WO-46: Preserve process authority for shell observations
2
+
3
+ Status: repository repair complete and delivered to `origin/main`. Publication and live validation remain with the user.
4
+
5
+ ## Finding and root repair
6
+
7
+ Field review F1 recorded successful source reads changed into failures because source text contained `throw new Error`. The runner's semantic stdout scanner and ShellTool's own soft-failure scanner both infer process failure from arbitrary output. A controlled reproduction also finds a false failure after `cd . && cat` prints a documented error line.
8
+
9
+ Preserve observed command/process outcomes through both execution loops. Reconcile actual nonzero exits, timeout, primary-command failures and properly bound verifier protocol results before workflow settlement. Source or documentation reads must not manufacture runtime or verification failures from their bytes. Share existing command semantics where needed to bind explicit verifier output; retain existing import compatibility.
10
+
11
+ ## Owned paths
12
+
13
+ - `packages/orchestrator/src/agenticRunner.ts`: shared tool result reconciliation and removal of generic semantic failure authority.
14
+ - `packages/execution/src/tools/shell.ts`: process/result producer and explicit verifier binding.
15
+ - Existing command-classification modules and compatibility exports, if required for one shared interpretation.
16
+ - Shell and actual-runner regression tests for both loops.
17
+
18
+ ## Acceptance
19
+
20
+ - [x] Successful source reads containing exception, compiler, test and verifier-marker literals remain successful observations.
21
+ - [x] Real nonzero, timeout, masked pipeline primary failures and bound explicit verification failures remain failed.
22
+ - [x] Both production loops and workflow settlement observe the same typed result.
23
+ - [x] No source output is promoted into verification authority; existing task/shell regressions pass.
24
+ - [x] Record full package verification, scoped commit and push.
25
+
26
+ ## Verification evidence
27
+
28
+ - 192 focused orchestrator tests passed, including both execution loops, actual ShellTool reads and verifier controls. Log: `/tmp/omnius-wo46-focused-final.log`.
29
+ - 63 focused execution tests passed; the complete execution package then passed 1,820 tests with 3 pre-existing skips across 163 suites. Log: `/tmp/omnius-remedies-execution-full.log`.
30
+ - Execution build, orchestrator typecheck and whitespace checks passed.
31
+ - Terminal-report handling of accepted negative searches and preview exits is completed with WO48, using the same command classification. Publication/live validation remain pending with the user.
32
+
33
+ ## Repository acceptance and delivery
34
+
35
+ - Source repair: `a3c26eca`, delivered to `origin/main`; coordinated changes: `672bef53 (terminal observation/report integration)`.
36
+ - Clean rebuild of all workspace packages and final orchestrator rebuild passed.
37
+ - Execution: 1,820 passed / 3 existing skips across 163 suites.
38
+ - CLI: 2,575 passed across 274 suites. Rerun with `--maxWorkers=4 --minWorkers=1` cleared the original five UI import timeouts without changing those tests or their limits.
39
+ - Orchestrator: full run covered 228 suites with 2,847 passed / 1 existing skip / 1 outdated fixture failure. The fixture lacked required `pipefail`; it was corrected, and all 10 tests in that suite then passed. No production source changed after the full run began. This is full-suite coverage plus the corrected suite rerun, not a claim that the initial command exited successfully.
40
+ - Across these packages, all 7,243 distinct non-skipped tests have a passing result on the final implementation. Logs: `/tmp/omnius-remedies-execution-full.log`, `/tmp/omnius-remedies-orchestrator-full.log`, `/tmp/omnius-remedies-receipt-fixture.log`, `/tmp/omnius-remedies-cli-final.log`.
41
+
42
+ Checks used hermetic backend/Telegram transport and synthetic local process/project fixtures. No live inference, service restart, installed-package replacement, Telegram sends or publication were performed. The existing publish staging and unrelated discovery changes were preserved.
@@ -0,0 +1,54 @@
1
+ # WO-47: Validate and execute diagnostics without blocking the agent
2
+
3
+ Status: repository repair complete and delivered to `origin/main`. Publication and live validation remain with the user.
4
+
5
+ ## Finding and root repair
6
+
7
+ Field review F2 recorded `requestedSteps.filter is not a function`: a JSON-encoded string reached an unchecked array cast. Review also found silently omitted requested checks, successful 0/0 results, implicit package downloads, and synchronous subprocess calls that can block typing and Stop for two minutes per step.
8
+
9
+ Use an executable input contract at runner admission and direct invocation. Validate requested steps and project paths before launching anything. Report unavailable requested checks and empty detection explicitly. Use configured local commands through the existing asynchronous process runner, with bounded output, timeout and cancellation of the owning process group. Preserve actual per-step outcomes in typed command receipts; never invent an aggregate successful command.
10
+
11
+ ## Owned paths
12
+
13
+ - `packages/execution/src/tools/diagnostic.ts` and diagnostic tests.
14
+ - `packages/execution/src/types.ts`: optional array of actual command receipts.
15
+ - Runner and CLI adapter receipt propagation, coordinated with WO-48.
16
+
17
+ ## Acceptance
18
+
19
+ - [x] String/null/object/invalid/empty steps and invalid path/fix arguments are rejected before execution, with actionable errors.
20
+ - [x] Valid subsets execute exactly; explicit unavailable checks and zero detected checks cannot report success.
21
+ - [x] No implicit dependency download or empty-test override; actual exit failures remain visible regardless of stdout.
22
+ - [x] Event-loop responsiveness, bounded timeout, Stop, descendant cleanup and sibling isolation are exercised with local fixtures.
23
+ - [x] Actual per-step receipts survive adapter, runner evidence recording and terminal reporting, including mixed outcomes.
24
+ - [x] Record full package verification, scoped commit and push.
25
+
26
+ ## Implemented behavior and focused verification
27
+
28
+ `DiagnosticTool.validateInput` and direct `execute` share the same value-level guard. The production CLI regression reproduces the field's JSON-string `steps` through the real adapter and runner: validation rejects it before the tool body or a process executes. Unsupported or unavailable explicit checks reject the whole request before dispatch. Default detection requires at least one available command.
29
+
30
+ Configured package scripts run through `npm run <step>` with argument arrays; fallback runners must already exist in a local `node_modules/.bin` directory. The tool no longer launches `npx` or adds Jest's `--passWithNoTests`. Local ESLint and Biome receive their respective `--fix` and `--write` flags. Commands execute using `runProcessBuffer` with a copied environment, bounded capture, a per-command timeout, and private process-group cancellation on Unix. Stop drains the owning invocation and prevents later steps; another tool instance remains independent. Build, lint fixes and package scripts retain external-effect authority.
31
+
32
+ `ToolResult.executionReceipts` carries one actual command receipt per attempted step. Completed processes retain their actual exit status. Timeout, cancellation and incomplete capture use null effective/primary status, preserving any raw zero exit in `runnerExitCode`; a cancelled process cannot become verification success merely by handling TERM with `exit(0)`. Earlier completed checks remain separately attributable when a later check fails. No aggregate successful command is invented.
33
+
34
+ Focused verification (2026-09-05, synthetic processes and temporary projects only):
35
+
36
+ - `packages/execution/tests/diagnostic.test.ts`: 35 passing tests.
37
+ - Existing `packages/execution/tests/process-async.test.ts`: 1 passing descendant-cancellation test.
38
+ - `packages/cli/tests/diagnostic-boundary.test.ts`: 3 passing production-boundary tests, including exact two-command terminal reports with exit pairs `[0, 0]` and `[0, 7]`.
39
+ - Execution typecheck passed. Logs: `/tmp/omnius-wo47-diagnostic-tests.log`, `/tmp/omnius-wo47-cli-boundary-tests.log`, `/tmp/omnius-wo47-execution-types.log`.
40
+
41
+ These receipts prove the observed process outcomes, not the completeness of a project's test suite. Configured package scripts remain project-owned commands and may themselves perform network access or modify files; the diagnostic tool does not invent a sandbox or infer an exact changed-file list. Unix descendant ownership is covered on the current host. Publication and live validation remain outside this workorder's repair execution.
42
+
43
+ The complete execution package passed 1,820 tests with 3 pre-existing skips across 163 suites (`/tmp/omnius-remedies-execution-full.log`). The production CLI boundary tests and receipt consumer ship with WO48 so that each source commit has its dependencies.
44
+
45
+ ## Repository acceptance and delivery
46
+
47
+ - Source repair: `d4c37489`, delivered to `origin/main`; coordinated changes: `672bef53 (CLI adapter and report integration)`.
48
+ - Clean rebuild of all workspace packages and final orchestrator rebuild passed.
49
+ - Execution: 1,820 passed / 3 existing skips across 163 suites.
50
+ - CLI: 2,575 passed across 274 suites. Rerun with `--maxWorkers=4 --minWorkers=1` cleared the original five UI import timeouts without changing those tests or their limits.
51
+ - Orchestrator: full run covered 228 suites with 2,847 passed / 1 existing skip / 1 outdated fixture failure. The fixture lacked required `pipefail`; it was corrected, and all 10 tests in that suite then passed. No production source changed after the full run began. This is full-suite coverage plus the corrected suite rerun, not a claim that the initial command exited successfully.
52
+ - Across these packages, all 7,243 distinct non-skipped tests have a passing result on the final implementation. Logs: `/tmp/omnius-remedies-execution-full.log`, `/tmp/omnius-remedies-orchestrator-full.log`, `/tmp/omnius-remedies-receipt-fixture.log`, `/tmp/omnius-remedies-cli-final.log`.
53
+
54
+ Checks used hermetic backend/Telegram transport and synthetic local process/project fixtures. No live inference, service restart, installed-package replacement, Telegram sends or publication were performed. The existing publish staging and unrelated discovery changes were preserved.
@@ -0,0 +1,63 @@
1
+ # WO-48: Retain command evidence in conversation results
2
+
3
+ Status: repository repair complete and delivered to `origin/main`. Publication and live validation remain with the user.
4
+
5
+ ## Finding and root repair
6
+
7
+ Field review F3 found an accepted conversation final with actual check output but an empty structured terminal report. Conversation mode intentionally disables the task ledger; the shared result producer therefore discards evidence before reporting.
8
+
9
+ Maintain report-only evidence for the exact conversation run and task epoch, reusing existing evidence types and immutable artifact storage. Preserve accepted process facts and observations independently of the task ledger. Persist the report snapshot before terminal delivery. Task claims, todos, workboards, mission artifacts and completion requirements must remain governed by their existing authority; enabling reporting cannot grant task or mutation permissions.
10
+
11
+ ## Owned paths
12
+
13
+ - `packages/orchestrator/src/agenticRunner.ts`: evidence producer, run/epoch initialization and terminal commit.
14
+ - `packages/orchestrator/src/conversationReportEvidence.ts` and production-runner/recovery tests.
15
+ - `packages/orchestrator/src/commandEvidenceReceipt.ts`: shared strict receipt parser before nested field access or recovery.
16
+ - `packages/cli/src/tui/tool-adapter.ts`: multiple command receipt preservation.
17
+ - `packages/orchestrator/src/terminalTaskReport.ts`: typed visible limits and successful observation outcomes.
18
+ - Actual mocked Telegram conversation/final-selection tests and adapter tests.
19
+
20
+ ## Acceptance
21
+
22
+ - [x] Actual conversation-mode execution retains typed command and observation evidence in both loops.
23
+ - [x] Actual Telegram chat route delivers a useful report when no separately authored final exists, including failed checks.
24
+ - [x] Task ledger/claims/todos/mission state remain absent; reporting does not change completion authority or tool permissions.
25
+ - [x] New runs, epoch replacement, stale callbacks and scope mismatch cannot inherit earlier report evidence.
26
+ - [x] Immutable persistence/recovery retains exact scoped evidence or reports an explicit gap; no fabricated restored facts.
27
+ - [x] Multi-step diagnostic outcomes are recorded individually, preserving earlier passes and later failure/cancellation.
28
+ - [x] Record full package verification, scoped commit and push.
29
+
30
+ ## Integration design and additional defects found
31
+
32
+ Conversation reporting retains successful reads/searches and typed command outcomes without creating the task completion ledger. Report facts never enter task admission, claim reconciliation or completion readiness. An assessment can complete while truthfully reporting a failed project check. Run and epoch initialization clear earlier evidence, and each tool dispatch captures its report identity before any await so late callbacks cannot populate a replacement report.
33
+
34
+ Multi-command results retain separate process outcomes. The aggregate keeps its own mutation/delivery success; an earlier passing child command cannot make a failed aggregate delivery or mutation successful. The CLI adapter carries every receipt through direct and streaming execution.
35
+
36
+ Actual Telegram regressions exposed a second presentation path: verification-audit gap prose included the tool's raw stdout summary. The terminal report now derives visible gap details from typed commands, paths and stable gap categories. Raw evidence remains inspectable. Successful negative searches and normalized read-preview SIGPIPE outcomes are observations with their original exits retained, rather than invented failures or verification passes.
37
+
38
+ Recovery storage uses immutable linked chunks plus a scoped selector, rather than rewriting the entire evidence history after every tool. Only new evidence is stored, avoiding quadratic use of the shared artifact-store quota on long runs. Recovery validates scope, ordering, identities, receipt shape and integrity before reconstructing report-only facts; it cannot deserialize task authority. Missing or rejected evidence becomes an explicit report limit. Final delivery requires the current evidence to be durably persisted.
39
+
40
+ ## Focused verification
41
+
42
+ - 16 production-runner cases passed, covering both loops; mixed pass/failure/timeout; no task authority; new-run isolation; stale callbacks after epoch replacement; read/search observations; separate aggregate effects; malformed receipt fields with valid siblings; repeated identical attempts; and exact interrupted recovery or an explicit gap without replaying checks. Log: `/tmp/omnius-wo48-runner.log`.
43
+ - 49 helper cases passed, including 100 receipts under a 256 KiB storage budget, immutable old-head recovery, malformed scope/receipt/task-authority rejection, corruption and missing ancestors, count/byte/cycle limits, and publication fault points. The three recovery-limit cases additionally prove that rejected count/byte bounds prevent CAS body reads. Logs: `/tmp/omnius-wo48-report-evidence-helper-tests.log`, `/tmp/omnius-wo48-report-recovery-limits-tests.log`.
44
+ - 36 terminal report tests passed. Log: `/tmp/omnius-wo48-terminal-report.log`.
45
+ - Six actual Telegram conversation cases, three adapter paths and three actual diagnostic-to-runner cases passed in the complete CLI suite. Final CLI: 2,575 passing tests across 274 suites with four workers; original unlimited concurrency hit five existing UI import timeouts. Log: `/tmp/omnius-remedies-cli-final.log`.
46
+ - Clean all-workspace build and final orchestrator rebuild passed. Logs: `/tmp/omnius-remedies-clean-build.log`, `/tmp/omnius-remedies-orchestrator-final-build.log`.
47
+
48
+ The strict receipt parser also exposed an older valid-verification mock missing required `pipefail`; the fixture now supplies the actual contract field, and all ten observed-authority cases pass without weakening readiness assertions (`/tmp/omnius-remedies-receipt-fixture.log`).
49
+
50
+ The review additionally reproduced invalid nested receipt fields throwing before validation, and an incomplete aggregate disappearing behind its passed prerequisite. The recorder now validates before nested field access, preserves unverified evidence and valid siblings, and records the aggregate failure separately. Separate identical array receipts remain distinct executions; only a legacy singleton alias is deduplicated.
51
+
52
+ Recovery is bounded by configured CAS and host recovery limits; it is not unlimited retention or a filesystem-effect transaction. An interruption before a tool observation is published still uses the existing interruption reconciliation. Publication and live validation remain with the user.
53
+
54
+ ## Repository acceptance and delivery
55
+
56
+ - Source repair: `672bef53`, delivered to `origin/main`; coordinated changes: `a3c26eca and d4c37489 (shared command semantics and diagnostic receipts)`.
57
+ - Clean rebuild of all workspace packages and final orchestrator rebuild passed.
58
+ - Execution: 1,820 passed / 3 existing skips across 163 suites.
59
+ - CLI: 2,575 passed across 274 suites. Rerun with `--maxWorkers=4 --minWorkers=1` cleared the original five UI import timeouts without changing those tests or their limits.
60
+ - Orchestrator: full run covered 228 suites with 2,847 passed / 1 existing skip / 1 outdated fixture failure. The fixture lacked required `pipefail`; it was corrected, and all 10 tests in that suite then passed. No production source changed after the full run began. This is full-suite coverage plus the corrected suite rerun, not a claim that the initial command exited successfully.
61
+ - Across these packages, all 7,243 distinct non-skipped tests have a passing result on the final implementation. Logs: `/tmp/omnius-remedies-execution-full.log`, `/tmp/omnius-remedies-orchestrator-full.log`, `/tmp/omnius-remedies-receipt-fixture.log`, `/tmp/omnius-remedies-cli-final.log`.
62
+
63
+ Checks used hermetic backend/Telegram transport and synthetic local process/project fixtures. No live inference, service restart, installed-package replacement, Telegram sends or publication were performed. The existing publish staging and unrelated discovery changes were preserved.
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "omnius",
3
- "version": "1.0.697",
3
+ "version": "1.0.698",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "omnius",
9
- "version": "1.0.697",
9
+ "version": "1.0.698",
10
10
  "bundleDependencies": [
11
11
  "image-to-ascii"
12
12
  ],
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "omnius",
3
- "version": "1.0.697",
3
+ "version": "1.0.698",
4
4
  "description": "AI coding agent powered by open-source models (Ollama/vLLM) — interactive TUI with agentic tool-calling loop",
5
5
  "type": "module",
6
6
  "main": "./dist/library.js",