newmark-agent 0.6.4 → 0.6.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/config.example.json +13 -3
  2. package/dist/cli-commands.d.ts +1 -1
  3. package/dist/cli-commands.js +54 -3
  4. package/dist/cli-discovery.js +3 -0
  5. package/dist/conversation-utility-host.bundle.cjs +4472 -1767
  6. package/dist/conversation-utility-host.js +94 -7
  7. package/dist/core/agent.d.ts +126 -12
  8. package/dist/core/agent.js +785 -172
  9. package/dist/core/agentKernelRunner.d.ts +7 -1
  10. package/dist/core/agentKernelRunner.js +160 -20
  11. package/dist/core/autoRouter.d.ts +49 -44
  12. package/dist/core/autoRouter.js +117 -302
  13. package/dist/core/computerUseSession.js +1 -1
  14. package/dist/core/config.js +27 -11
  15. package/dist/core/continuation/contracts.d.ts +1 -1
  16. package/dist/core/continuation/store.d.ts +91 -1
  17. package/dist/core/continuation/store.js +291 -61
  18. package/dist/core/conversationKernel.d.ts +49 -7
  19. package/dist/core/conversationKernel.js +303 -51
  20. package/dist/core/conversationStateDocument.d.ts +18 -0
  21. package/dist/core/conversationStateDocument.js +88 -0
  22. package/dist/core/conversationVisualBudget.d.ts +74 -0
  23. package/dist/core/conversationVisualBudget.js +153 -0
  24. package/dist/core/electronUtilityAgentClient.d.ts +3 -2
  25. package/dist/core/electronUtilityAgentClient.js +32 -8
  26. package/dist/core/electronUtilityRuntimePool.d.ts +28 -3
  27. package/dist/core/electronUtilityRuntimePool.js +93 -24
  28. package/dist/core/fileWriteObservation.d.ts +41 -0
  29. package/dist/core/fileWriteObservation.js +454 -0
  30. package/dist/core/hostRuntimeHooks.d.ts +23 -0
  31. package/dist/core/hostRuntimeHooks.js +17 -0
  32. package/dist/core/jevDecision.d.ts +144 -0
  33. package/dist/core/jevDecision.js +226 -0
  34. package/dist/core/memoryProbe.d.ts +24 -0
  35. package/dist/core/memoryProbe.js +104 -0
  36. package/dist/core/mobilePairing.d.ts +7 -1
  37. package/dist/core/mobilePairing.js +9 -1
  38. package/dist/core/performanceDiagnostics.d.ts +1 -1
  39. package/dist/core/routeDecisionValidator.d.ts +86 -0
  40. package/dist/core/routeDecisionValidator.js +249 -0
  41. package/dist/core/routeEligibility.d.ts +52 -0
  42. package/dist/core/routeEligibility.js +108 -0
  43. package/dist/core/runtimeMemoryBudget.d.ts +6 -0
  44. package/dist/core/runtimeMemoryBudget.js +12 -0
  45. package/dist/core/toolPolicy.d.ts +1 -1
  46. package/dist/core/toolPolicy.js +1 -1
  47. package/dist/core/utilityAgentProtocol.d.ts +8 -10
  48. package/dist/core/utilityHostToolRouter.js +2 -0
  49. package/dist/core/visualDownscale.d.ts +6 -0
  50. package/dist/core/visualDownscale.js +137 -0
  51. package/dist/core/wslAgentClient.d.ts +1 -2
  52. package/dist/core/wslAgentClient.js +0 -8
  53. package/dist/core/wslAgentProtocol.d.ts +1 -10
  54. package/dist/core/wslAgentRuntimePool.d.ts +1 -3
  55. package/dist/core/wslAgentRuntimePool.js +17 -22
  56. package/dist/llm/provider.d.ts +3 -1
  57. package/dist/llm/provider.js +53 -9
  58. package/dist/main.js +107 -30
  59. package/dist/preload.js +3 -1
  60. package/dist/server.js +30 -18
  61. package/dist/tools/computerUse.d.ts +6 -0
  62. package/dist/tools/computerUse.js +406 -21
  63. package/dist/tools/computerUsePowerShellHost.d.ts +6 -1
  64. package/dist/tools/computerUsePowerShellHost.js +105 -31
  65. package/dist/tools/index.d.ts +3 -4
  66. package/dist/tools/index.js +146 -66
  67. package/dist/tui/src/adapters/core-runtime-adapter.js +13 -1
  68. package/dist/tui/src/i18n.js +12 -1
  69. package/dist/tui/src/render.js +9 -2
  70. package/dist/tui/src/settings-schema.js +19 -0
  71. package/dist/tui/src/state.js +69 -1
  72. package/dist/ui/index.html +651 -233
  73. package/dist/ui/lucide-sprite.svg +0 -8
  74. package/dist/wsl-agent-host.bundle.cjs +4302 -1767
  75. package/dist/wsl-agent-host.js +0 -3
  76. package/package.json +39 -10
@@ -7,6 +7,8 @@ const conversationTarget_1 = require("./core/conversationTarget");
7
7
  const utilityHostToolBridge_1 = require("./core/utilityHostToolBridge");
8
8
  const terminalTakeover_1 = require("./tools/terminalTakeover");
9
9
  const runtimeLifecycle_1 = require("./core/runtimeLifecycle");
10
+ const memoryProbe_1 = require("./core/memoryProbe");
11
+ const workEventCoalescer_1 = require("./core/workEventCoalescer");
10
12
  const root = String(process.env.NEWMARK_RUNTIME_ROOT || '');
11
13
  const expectedRuntimeKey = String(process.env.NEWMARK_RUNTIME_KEY || '');
12
14
  if (!root)
@@ -17,6 +19,11 @@ const parentPort = process.parentPort;
17
19
  if (!parentPort)
18
20
  throw new Error('Electron utility parentPort is unavailable');
19
21
  const host = new agent_1.Agent(root, { runtimeLifecycleRole: 'utility' });
22
+ /**
23
+ * Diagnostic probe (source-only, env-gated). `memoryProbe.disabled.ts` replaces
24
+ * this module in packaged builds, so no probe code ships in the final artifact.
25
+ */
26
+ const hostHooks = (0, memoryProbe_1.installHostRuntimeHooks)({ root });
20
27
  host.tools.setHostProfile({
21
28
  kind: 'electron-utility',
22
29
  platform: process.platform,
@@ -26,9 +33,35 @@ host.tools.setHostProfile({
26
33
  const kernel = new conversationKernel_1.ConversationKernel(root, host, null);
27
34
  let activeTarget = null;
28
35
  function post(value) {
29
- parentPort.postMessage(value);
36
+ hostHooks.countPosted(value);
37
+ try {
38
+ parentPort.postMessage(value);
39
+ }
40
+ catch { /* the parent may already be gone during shutdown */ }
41
+ }
42
+ /**
43
+ * dev-0.6.6 long-run stability: coalesce response/thought deltas **inside the
44
+ * host** before they cross the process boundary.
45
+ *
46
+ * Measured defect: the host posted one IPC message per model delta (18,746
47
+ * messages / 722 MiB for ~31 MB of model text — a 23x amplification) and the
48
+ * heap snapshot showed 10,728 retained per-delta strings. The main process
49
+ * already coalesces for the renderer (`main.ts` WorkEventCoalescer) and the
50
+ * renderer expands `coalescedDeltas` back into individual deltas, so batching at
51
+ * the host boundary changes nothing the user or the transcript sees — it only
52
+ * stops the per-delta pile-up. Lifecycle/tool events still flush immediately and
53
+ * keep their ordering.
54
+ */
55
+ const hostWorkCoalescer = new workEventCoalescer_1.WorkEventCoalescer(event => post({ event: 'work', data: event }), Math.max(4, Number(process.env.NEWMARK_HOST_COALESCE_MS) || 16));
56
+ function flushHostWorkEvents() {
57
+ try {
58
+ hostWorkCoalescer.flushAll();
59
+ }
60
+ catch { /* shutdown best effort */ }
30
61
  }
31
- kernel.subscribe(event => post({ event: 'work', data: event }));
62
+ process.on('exit', flushHostWorkEvents);
63
+ process.on('beforeExit', flushHostWorkEvents);
64
+ kernel.subscribe(event => hostWorkCoalescer.push(event));
32
65
  function checkedTarget(target) {
33
66
  const normalized = (0, conversationTarget_1.normalizeConversationTarget)(target);
34
67
  if (normalized.runtimeKey !== expectedRuntimeKey) {
@@ -37,6 +70,61 @@ function checkedTarget(target) {
37
70
  activeTarget = normalized;
38
71
  return normalized;
39
72
  }
73
+ /**
74
+ * dev-0.6.6 long-run memory reclamation: the transcript part of a snapshot is
75
+ * only sent when the caller cannot already prove it holds that exact revision.
76
+ *
77
+ * Measured before this: RPC responses (mostly snapshots) were **86% of all
78
+ * cross-process bytes** in a 12-round load (24 responses = 329 MiB, single
79
+ * response up to 38 MB) because every poll re-serialized the whole conversation.
80
+ *
81
+ * The handshake is caller-driven: a slim payload is only produced when the
82
+ * caller supplies the revision it currently holds. A caller that holds nothing
83
+ * (or painted another conversation) receives the full transcript; the main
84
+ * process only attaches a revision it received from that same renderer through
85
+ * `conversation:ackTranscript`.
86
+ */
87
+ const transcriptRevisions = new Map();
88
+ function transcriptRevisionOf(snapshot) {
89
+ const messages = Array.isArray(snapshot.chatMessages) ? snapshot.chatMessages : [];
90
+ const runs = Array.isArray(snapshot.workRuns) ? snapshot.workRuns : [];
91
+ const last = messages.length ? messages[messages.length - 1] : null;
92
+ const lastRun = runs.length ? runs[runs.length - 1] : null;
93
+ // Only what the renderer actually paints takes part in the revision: the
94
+ // visible message window and the work-run summary. `history` deliberately
95
+ // does NOT: it grows on every tool round inside a turn, and folding it in made
96
+ // every mid-turn poll look "changed" (measured: no slim snapshots at all).
97
+ // Anything derived from `history` is only recomputed when the window changes,
98
+ // which is exactly when a full payload is sent anyway.
99
+ return [
100
+ messages.length,
101
+ String(last?.messageId || last?.clientMessageId || last?.timestamp || ''),
102
+ String(typeof last?.content === 'string' ? last.content.length : ''),
103
+ String(snapshot.totalMessages ?? ''),
104
+ String(snapshot.windowStart ?? ''),
105
+ runs.length,
106
+ String(lastRun?.status || ''),
107
+ String(lastRun?.endedAt || ''),
108
+ ].join('|');
109
+ }
110
+ function applyTranscriptPolicy(snapshot, options) {
111
+ const revision = transcriptRevisionOf(snapshot);
112
+ const key = `${snapshot.target?.workspaceId || ''}::${snapshot.target?.conversationId || ''}`;
113
+ transcriptRevisions.set(key, revision);
114
+ const callerRevision = String(options.transcriptRevision || '');
115
+ const full = options.transcript === 'force' || !callerRevision || callerRevision !== revision;
116
+ if (full)
117
+ return { ...snapshot, transcriptRevision: revision };
118
+ return {
119
+ ...snapshot,
120
+ chatMessages: undefined,
121
+ history: undefined,
122
+ workRuns: undefined,
123
+ workEvents: undefined,
124
+ transcriptUnchanged: true,
125
+ transcriptRevision: revision,
126
+ };
127
+ }
40
128
  (0, utilityHostToolBridge_1.configureUtilityHostToolBridge)(request => post({ event: 'host_tool_request', data: request }), () => activeTarget ? {
41
129
  workspaceId: activeTarget.workspaceId,
42
130
  conversationId: activeTarget.conversationId,
@@ -64,8 +152,10 @@ async function handle(request) {
64
152
  const result = await kernel.prompt(request.params.message, target, request.params.options, request.params.queueMode);
65
153
  return { ...result, backend: 'utility', pid: process.pid };
66
154
  }
67
- if (request.method === 'snapshot')
68
- return kernel.snapshot(checkedTarget(request.params.target), request.params.options);
155
+ if (request.method === 'snapshot') {
156
+ const options = request.params.options || {};
157
+ return applyTranscriptPolicy(kernel.snapshot(checkedTarget(request.params.target), options), options);
158
+ }
69
159
  if (request.method === 'rewind') {
70
160
  return kernel.rewind(checkedTarget(request.params.target), request.params.messageIndex);
71
161
  }
@@ -87,9 +177,6 @@ async function handle(request) {
87
177
  if (request.method === 'context_compress') {
88
178
  return kernel.compressContext(checkedTarget(request.params.target), request.params.options);
89
179
  }
90
- if (request.method === 'rate_auto_route') {
91
- return kernel.rateAutoRoute(checkedTarget(request.params.target), request.params.score, request.params.routeId);
92
- }
93
180
  if (request.method === 'set_work_run_expanded') {
94
181
  return kernel.setWorkRunExpanded(checkedTarget(request.params.target), request.params.runId, request.params.expanded);
95
182
  }
@@ -55,11 +55,27 @@ export interface ModelValidationProgress {
55
55
  check: string;
56
56
  }>;
57
57
  }
58
- export interface AutoRouteRatingResult {
58
+ /** Editable settings for model-backed Auto decisions. */
59
+ export interface JevDecisionModelPatch {
60
+ enabled?: boolean;
61
+ timeoutMs?: number;
62
+ }
63
+ export interface JevDecisionModelState {
64
+ /** `models.auto_switch`: the user's Auto master switch. */
65
+ autoEnabled: boolean;
66
+ /** True when Auto is off: the decision model cannot be edited and cannot move a model. */
67
+ locked: boolean;
68
+ lockReason?: 'auto_disabled';
69
+ enabled: boolean;
70
+ decisionModel: 'current_eligible_model';
71
+ thinkingMode: 'off';
72
+ outputFormat: 'strict_json_schema';
73
+ timeoutMs: number;
74
+ }
75
+ export interface JevDecisionModelUpdateResult {
59
76
  ok: boolean;
60
- score?: -1 | 1;
61
- routeId?: string;
62
- reason?: 'invalid_score' | 'no_active_auto_route' | 'stale_route' | 'already_rated';
77
+ reason?: 'auto_disabled' | 'invalid_value';
78
+ state: JevDecisionModelState;
63
79
  }
64
80
  export declare const ROOT_AGENT_ACTOR_ID = "00000000-0000-4000-8000-000000000001";
65
81
  export declare function normalizeIntelligenceTier(value: unknown): 'low' | 'medium' | 'high' | 'xhigh' | 'max' | 'ultra';
@@ -415,6 +431,8 @@ export declare class Agent {
415
431
  private modelSwitchCompressionPromise;
416
432
  private activePeerAgents;
417
433
  private peerProviderCaches;
434
+ /** Reuses the provider connection for current-model JEV calls without sharing conversation build state. */
435
+ private readonly jevProviderCache;
418
436
  private awaitingAgentKernelRuntime;
419
437
  private pendingAgentKernelQueue;
420
438
  private linkedPlanAccess;
@@ -434,7 +452,7 @@ export declare class Agent {
434
452
  private finalizingWorkRunId;
435
453
  private agentRunService;
436
454
  private agentRunByWorkRunId;
437
- private readonly autoRouter;
455
+ private readonly routeController;
438
456
  private resolvedDeployment;
439
457
  private fixedDeployment;
440
458
  private preferredConversationModelSelection;
@@ -446,11 +464,21 @@ export declare class Agent {
446
464
  private lastRouteRetryDelayMs;
447
465
  private routeAttemptStartedAt;
448
466
  private readonly providerBalanceBlockedUntilByDeployment;
449
- private readonly explicitlyRatedRoutes;
450
467
  private conversationStateCache;
451
468
  private conversationStateCacheFingerprint;
469
+ private conversationStateWrittenFingerprint;
470
+ private conversationShardFingerprint;
471
+ private hydratedShardEntries;
472
+ /**
473
+ * Revision of the durable queue snapshot this runtime is working from. The
474
+ * merge only accepts a writer's queue when that writer has seen the persisted
475
+ * revision, so an outdated in-memory snapshot can neither drop queued work
476
+ * nor resurrect an entry that was already consumed or cleared.
477
+ */
478
+ private continuationRevision;
452
479
  private conversationStateFlushTimers;
453
480
  private conversationStateDirty;
481
+ private conversationStateLastWriteAt;
454
482
  private systemPromptCache;
455
483
  private toolDefinitionCache;
456
484
  private modelValidationPromise;
@@ -551,12 +579,65 @@ export declare class Agent {
551
579
  recordRouteSuccess(latencyMs?: number, throughput?: number): void;
552
580
  recordRouteToolOutcome(valid: boolean): void;
553
581
  updateProviders(value: unknown): void;
554
- clearLearnedModelPreferences(): void;
555
- rateActiveAutoRoute(score: number, expectedRouteId?: string): AutoRouteRatingResult;
556
- recordObjectiveRouteResult(success: boolean): void;
557
- private recordRouteQualityObservation;
582
+ /**
583
+ * Auto master switch. Every automatic model decision in this runtime is
584
+ * gated on it: when it is off the Agent neither routes nor edits the current
585
+ * model decision settings, and no code path may move the active model on its own.
586
+ */
587
+ autoDecisionEnabled(): boolean;
588
+ /** Current-model JEV settings exposed to GUI / TUI / CLI editors. */
589
+ jevDecisionModelState(): JevDecisionModelState;
590
+ /**
591
+ * Edit current-model Auto decision settings.
592
+ *
593
+ * Refuses every write while Auto is off (`auto_disabled`): with automatic
594
+ * decision-making disabled, not even the decision-model configuration may
595
+ * change — that is what "Auto off never moves a model" means at the editor
596
+ * boundary. Invalid values are rejected without touching the stored config.
597
+ */
598
+ updateJevDecisionModel(patch?: JevDecisionModelPatch): JevDecisionModelUpdateResult;
558
599
  private workspaceConversationKey;
559
600
  private workspaceConversationStorePath;
601
+ /**
602
+ * Heavy conversation payload that lives in a per-conversation shard file
603
+ * instead of the workspace index document. Everything else (title, pinning,
604
+ * order, draft, mode, goal, runtime owner, counters) stays in the index so
605
+ * listing, sorting and cross-process merge never need to parse a transcript.
606
+ */
607
+ private static readonly CONVERSATION_SHARD_FIELDS;
608
+ private workspaceConversationShardDir;
609
+ private conversationShardFileName;
610
+ private conversationShardPath;
611
+ /**
612
+ * Hydrate one conversation entry from its shard on first access. The heavy
613
+ * fields are non-enumerable so `{ ...entry }` (and therefore the index
614
+ * document written back to disk) never pulls a transcript into memory, while
615
+ * every existing reader that asks for `entry.chatMessages` / `entry.history`
616
+ * keeps working unchanged.
617
+ */
618
+ private attachConversationShard;
619
+ /**
620
+ * Heavy payload carried in memory by this entry, or `null` when the entry is
621
+ * a shard-backed index record whose heavy fields were never touched. That
622
+ * distinction is what keeps unrelated conversations off the write path.
623
+ */
624
+ private conversationShardSnapshot;
625
+ /**
626
+ * Force one shard-backed entry into memory. Used only on the rare recovery
627
+ * paths that must compare against a payload written by another runtime.
628
+ */
629
+ private hydrateConversationEntry;
630
+ /**
631
+ * Durable queue merge.
632
+ *
633
+ * Queue entries are the user's pending work, so a runtime that never saw an
634
+ * entry must not delete it, while an entry this runtime actually consumed
635
+ * must not be resurrected. Identity-based union minus exactly the consumed
636
+ * ids gives both guarantees without needing a global revision clock.
637
+ */
638
+ private mergeConversationContinuations;
639
+ /** Index-document form of a conversation entry: metadata plus a shard pointer. */
640
+ private conversationIndexEntryForDisk;
560
641
  private workspaceConversationStateKey;
561
642
  private workspaceConversationStateKeyFor;
562
643
  private workspaceConversationPrefix;
@@ -593,6 +674,16 @@ export declare class Agent {
593
674
  handleImageDisplayWithDescription(args: string, signal?: AbortSignal): Promise<string>;
594
675
  /** Keep media bytes in one content-addressed asset and state.json as refs. */
595
676
  private conversationEntryForDisk;
677
+ /**
678
+ * Bound the live Build ledger.
679
+ *
680
+ * `normalizeWorkRuns()` already windows the persisted/served ledger to 120
681
+ * runs, but the live `this.workRuns` array kept every Build block of the
682
+ * conversation (measured: 307 runs / 4.6 MiB after 300 turns). The renderer
683
+ * only ever shows the newest window, so older terminal runs can be dropped
684
+ * from memory while the conversation continues.
685
+ */
686
+ private pruneLiveWorkRuns;
596
687
  private normalizeWorkRuns;
597
688
  private recoverPersistedWorkRuns;
598
689
  private publishConversationTitle;
@@ -697,7 +788,20 @@ export declare class Agent {
697
788
  private normalizeContinuations;
698
789
  private normalizeConversationPlan;
699
790
  private readStoredConversationState;
791
+ /** Durable replacement-style text write (temp file + fsync + rename). */
792
+ private writeAtomicTextFile;
700
793
  private writeStoredConversationState;
794
+ /**
795
+ * Coalesced conversation-state write.
796
+ *
797
+ * Mid-turn persistence used to rewrite the whole workspace document
798
+ * synchronously on every kernel event (measured: 5.33 full rewrites per turn
799
+ * with no tools). The document grows with the conversation, so that cost grew
800
+ * with the turn count. This path keeps durability at turn boundaries (the
801
+ * terminal save stays synchronous and cancels any pending timer) while
802
+ * throttling intermediate saves: the latest snapshot wins, and bursts inside
803
+ * one throttle window collapse into a single write.
804
+ */
701
805
  private scheduleStoredConversationState;
702
806
  flushWorkspaceConversationState(): void;
703
807
  getStoredFlowSuspension(conversationId?: string, ws?: WorkspaceInfo | null): FlowSuspensionRecord | null;
@@ -1117,10 +1221,20 @@ export declare class Agent {
1117
1221
  private autoSwitchSubset;
1118
1222
  private autoRouteCandidates;
1119
1223
  private persistRouteDecision;
1120
- private loadLearnedRouteFeedback;
1121
- private recordRouteFeedbackFor;
1122
1224
  allModelNames(): string[];
1123
1225
  evaluateAndSwitch(task: string, override?: AgentPromptMessage['routePolicy']): Promise<boolean>;
1226
+ /**
1227
+ * Auto Router v2 decision pipeline:
1228
+ *
1229
+ * RouteEligibilityGuard → JevDecisionEngine → RouteDecisionValidator → pin
1230
+ *
1231
+ * The current conversation model is the only Auto decision source. If it is
1232
+ * disabled, times out, crashes or returns something the guard cannot accept,
1233
+ * Auto stops without selecting another deployment.
1234
+ */
1235
+ private decideAutoRoute;
1236
+ /** Resolve JEV to the current eligible conversation model, never a second model slot. */
1237
+ private resolveCurrentModelJevEngine;
1124
1238
  shouldExposeToolInterface(): boolean;
1125
1239
  modelIsUnavailable(modelName: string): boolean;
1126
1240
  private failedDeploymentsThisRun;