omnius 1.0.646 → 1.0.647

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -628934,6 +628934,67 @@ var init_personality = __esm({
628934
628934
  }
628935
628935
  });
628936
628936
 
628937
+ // packages/orchestrator/dist/ste-communication-policy.js
628938
+ function ensureSteCommunicationPolicy(messages2) {
628939
+ const hasPolicy = messages2.some((message2) => message2.role === "system" && typeof message2.content === "string" && message2.content.includes(STE_POLICY_MARKER));
628940
+ if (hasPolicy)
628941
+ return [...messages2];
628942
+ return [
628943
+ {
628944
+ role: "system",
628945
+ content: STE_COMMUNICATION_POLICY
628946
+ },
628947
+ ...messages2
628948
+ ];
628949
+ }
628950
+ var STE_POLICY_MARKER, STE_COMMUNICATION_POLICY;
628951
+ var init_ste_communication_policy = __esm({
628952
+ "packages/orchestrator/dist/ste-communication-policy.js"() {
628953
+ "use strict";
628954
+ STE_POLICY_MARKER = "[ASD-STE100 ISSUE 9 COMMUNICATION POLICY]";
628955
+ STE_COMMUNICATION_POLICY = `${STE_POLICY_MARKER}
628956
+
628957
+ Use ASD-STE100 Issue 9 for all natural-language communication.
628958
+ Apply this policy to external replies and internal agent communication.
628959
+
628960
+ - Use approved words when possible.
628961
+ - Treat necessary project and domain terms as technical terms.
628962
+ - Use one term for each concept.
628963
+ - Use American English spelling.
628964
+ - Do not use contractions, slang, jargon, or idioms.
628965
+ - Use active voice.
628966
+ - Use passive voice only when the agent is unknown.
628967
+ - Use simple verb tenses.
628968
+ - Use direct verbs.
628969
+ - Avoid phrasal verbs.
628970
+ - Keep each instruction at 20 words or fewer.
628971
+ - Give one instruction in each sentence.
628972
+ - Put a condition before its instruction.
628973
+ - Keep each descriptive sentence at 25 words or fewer.
628974
+ - Keep one topic in each paragraph.
628975
+ - Keep each paragraph at six sentences or fewer.
628976
+ - Use a vertical list for complex information.
628977
+ - Use clear connecting words.
628978
+ - Do not use semicolons.
628979
+ - Identify the level of each safety risk.
628980
+ - Start each safety instruction with a clear command or condition.
628981
+ - Explain the possible result of each safety risk.
628982
+
628983
+ Preserve exact content in these items:
628984
+
628985
+ - Code and code blocks
628986
+ - JSON, schemas, keys, and API data
628987
+ - Commands, paths, identifiers, and formal names
628988
+ - User input and source quotations
628989
+ - Logs, tool output, and test output
628990
+ - Required machine formats
628991
+
628992
+ Use STE for all text that explains an exact item.
628993
+ If an exact format conflicts with STE, use the exact format.
628994
+ Do not claim certified ASD-STE100 conformance without a qualified human review.`;
628995
+ }
628996
+ });
628997
+
628937
628998
  // packages/orchestrator/dist/reference-contracts.js
628938
628999
  import { createHash as createHash37 } from "node:crypto";
628939
629000
  function compactSource(source, maxChars) {
@@ -648985,7 +649046,7 @@ function classifyThinkOutcome(raw) {
648985
649046
  }
648986
649047
  return null;
648987
649048
  }
648988
- var PRESERVE_MODEL_VISIBLE_HISTORY, TOOL_SUBSETS, TOOL_AUTO_DEMOTE_TURNS, LEGACY_ACTION_REASON_KEYS, CLAIM_TASK_STOPWORDS, SYSTEM_PROMPT, SYSTEM_PROMPT_MEDIUM, SYSTEM_PROMPT_SMALL, VISUAL_TOOLS, AUDIO_TOOLS, SOCIAL_TOOLS, SPATIAL_TOOLS, CODE_TOOLS, AgenticRunner, OllamaAgenticBackend;
649049
+ var PRESERVE_MODEL_VISIBLE_HISTORY, TOOL_SUBSETS, TOOL_AUTO_DEMOTE_TURNS, LEGACY_ACTION_REASON_KEYS, CLAIM_TASK_STOPWORDS, SYSTEM_PROMPT_COMMON, SYSTEM_PROMPT, SYSTEM_PROMPT_MEDIUM, SYSTEM_PROMPT_SMALL, VISUAL_TOOLS, AUDIO_TOOLS, SOCIAL_TOOLS, SPATIAL_TOOLS, CODE_TOOLS, AgenticRunner, OllamaAgenticBackend;
648989
649050
  var init_agenticRunner = __esm({
648990
649051
  "packages/orchestrator/dist/agenticRunner.js"() {
648991
649052
  "use strict";
@@ -649008,6 +649069,7 @@ var init_agenticRunner = __esm({
649008
649069
  init_ollama_pool();
649009
649070
  init_personality();
649010
649071
  init_promptLoader();
649072
+ init_ste_communication_policy();
649011
649073
  init_reference_contracts();
649012
649074
  init_phase_skill_guidance();
649013
649075
  init_deliveryCoverage();
@@ -649183,9 +649245,18 @@ var init_agenticRunner = __esm({
649183
649245
  "new",
649184
649246
  "old"
649185
649247
  ]);
649186
- SYSTEM_PROMPT = loadPrompt("agentic/system-large.md");
649187
- SYSTEM_PROMPT_MEDIUM = loadPrompt("agentic/system-medium.md");
649188
- SYSTEM_PROMPT_SMALL = loadPrompt("agentic/system-small.md");
649248
+ SYSTEM_PROMPT_COMMON = `${STE_COMMUNICATION_POLICY}
649249
+
649250
+ ${loadPrompt("agentic/system-common.md")}`;
649251
+ SYSTEM_PROMPT = `${SYSTEM_PROMPT_COMMON}
649252
+
649253
+ ${loadPrompt("agentic/system-large.md")}`;
649254
+ SYSTEM_PROMPT_MEDIUM = `${SYSTEM_PROMPT_COMMON}
649255
+
649256
+ ${loadPrompt("agentic/system-medium.md")}`;
649257
+ SYSTEM_PROMPT_SMALL = `${SYSTEM_PROMPT_COMMON}
649258
+
649259
+ ${loadPrompt("agentic/system-small.md")}`;
649189
649260
  VISUAL_TOOLS = /* @__PURE__ */ new Set([
649190
649261
  "vision",
649191
649262
  "camera_capture",
@@ -676242,6 +676313,10 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676242
676313
  return `${serviceBaseUrl.replace(/\/+$/, "")}${path16}`;
676243
676314
  }
676244
676315
  async chatCompletion(request) {
676316
+ request = {
676317
+ ...request,
676318
+ messages: ensureSteCommunicationPolicy(request.messages)
676319
+ };
676245
676320
  if (this._provider.protocol === "anthropic-messages") {
676246
676321
  return this._anthropicChatCompletion(request);
676247
676322
  }
@@ -676268,7 +676343,8 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676268
676343
  }
676269
676344
  const requestMessages2 = cleanedMessages;
676270
676345
  const responseFormat = request.responseFormat ?? request.response_format;
676271
- const isOllama = this._provider.protocol === "ollama" && shouldUseOllamaPoolForBaseUrl(this.baseUrl);
676346
+ const isOllamaTransport = this._provider.protocol === "ollama";
676347
+ const useOllamaPool = isOllamaTransport && shouldUseOllamaPoolForBaseUrl(this.baseUrl);
676272
676348
  const body = {
676273
676349
  model: this.model,
676274
676350
  messages: requestMessages2,
@@ -676276,20 +676352,20 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676276
676352
  temperature: request.temperature,
676277
676353
  max_tokens: effectiveMaxTokens
676278
676354
  };
676279
- if (isOllama) {
676355
+ if (isOllamaTransport) {
676280
676356
  body["think"] = effectiveThink;
676281
676357
  }
676282
676358
  if (responseFormat !== void 0) {
676283
676359
  body["response_format"] = responseFormat;
676284
676360
  }
676285
676361
  const reqNumCtx = request.numCtx ?? request.num_ctx;
676286
- if (isOllama && Number.isFinite(reqNumCtx) && (reqNumCtx ?? 0) > 0) {
676362
+ if (isOllamaTransport && Number.isFinite(reqNumCtx) && (reqNumCtx ?? 0) > 0) {
676287
676363
  const opts = body["options"] ?? {};
676288
676364
  opts["num_ctx"] = reqNumCtx;
676289
676365
  body["options"] = opts;
676290
676366
  body["num_ctx"] = reqNumCtx;
676291
676367
  }
676292
- let poolSlot = isOllama ? await getOllamaPool({ baseInstanceUrl: this.baseUrl }).acquire({
676368
+ let poolSlot = useOllamaPool ? await getOllamaPool({ baseInstanceUrl: this.baseUrl }).acquire({
676293
676369
  model: this.model,
676294
676370
  queuePolicy: request.poolQueuePolicy,
676295
676371
  queueTimeoutMs: request.poolQueueTimeoutMs
@@ -676333,7 +676409,7 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676333
676409
  ]);
676334
676410
  let nativePreferredAttempted = false;
676335
676411
  try {
676336
- if (isOllama && request.preferNativeOllamaChat === true) {
676412
+ if (isOllamaTransport && request.preferNativeOllamaChat === true) {
676337
676413
  nativePreferredAttempted = true;
676338
676414
  try {
676339
676415
  const native = await this.nativeOllamaChatCompletion(request, {
@@ -676406,7 +676482,7 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676406
676482
  const independentOutcome = effectiveThink !== true ? classifyThinkOutcome(responseText2) : null;
676407
676483
  const emptyIndependentOutcome = independentOutcome === "empty_after_strip" || independentOutcome === "unclosed_think";
676408
676484
  const expectsVisibleContent = responseFormat !== void 0 || (request.tools ?? []).length === 0;
676409
- const shouldProbeNativeOnEmpty = isOllama && !nativePreferredAttempted && expectsVisibleContent && emptyIndependentOutcome;
676485
+ const shouldProbeNativeOnEmpty = isOllamaTransport && !nativePreferredAttempted && expectsVisibleContent && emptyIndependentOutcome;
676410
676486
  const shouldRecoverFromEmpty = request.disableEmptyContentRecovery !== true && expectsVisibleContent && emptyIndependentOutcome;
676411
676487
  const justSuppressed = this._thinkSuppressed && this._thinkFailStreak === _OllamaAgenticBackend._thinkFailThreshold;
676412
676488
  const shouldRetryThinkGuard = outcome !== null && effectiveThink === true && (justSuppressed || outcome === "empty_after_strip" || outcome === "unclosed_think");
@@ -676443,12 +676519,12 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676443
676519
  temperature: request.temperature,
676444
676520
  max_tokens: request.maxTokens
676445
676521
  };
676446
- if (isOllama)
676522
+ if (isOllamaTransport)
676447
676523
  retryBody["think"] = false;
676448
676524
  if (responseFormat !== void 0 && shouldRetryThinkGuard && !shouldRecoverFromEmpty) {
676449
676525
  retryBody["response_format"] = responseFormat;
676450
676526
  }
676451
- if (isOllama && Number.isFinite(reqNumCtx) && (reqNumCtx ?? 0) > 0) {
676527
+ if (isOllamaTransport && Number.isFinite(reqNumCtx) && (reqNumCtx ?? 0) > 0) {
676452
676528
  const retryOptsBody = retryBody["options"] ?? {};
676453
676529
  retryOptsBody["num_ctx"] = reqNumCtx;
676454
676530
  retryBody["options"] = retryOptsBody;
@@ -676516,6 +676592,10 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676516
676592
  }
676517
676593
  }
676518
676594
  async nativeOllamaChatCompletion(request, transport = {}) {
676595
+ request = {
676596
+ ...request,
676597
+ messages: ensureSteCommunicationPolicy(request.messages)
676598
+ };
676519
676599
  const cleanedMessages = applyMemoryPrefixToMessages(normalizeMessagesForStrictOpenAI(sanitizeHistoryThink(request.messages, {
676520
676600
  preserveModelVisibleHistory: request[PRESERVE_MODEL_VISIBLE_HISTORY] === true
676521
676601
  })), request.memoryPrefix);
@@ -676628,6 +676708,10 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676628
676708
  * Ollama pool routing as non-stream completions.
676629
676709
  */
676630
676710
  async *chatCompletionStream(request) {
676711
+ request = {
676712
+ ...request,
676713
+ messages: ensureSteCommunicationPolicy(request.messages)
676714
+ };
676631
676715
  if (this._provider.protocol === "anthropic-messages") {
676632
676716
  const result = await this._anthropicChatCompletion(request);
676633
676717
  const message2 = result.choices[0]?.message;
@@ -676691,20 +676775,21 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676691
676775
  stream: true,
676692
676776
  stream_options: { include_usage: true }
676693
676777
  };
676694
- const isOllama = this._provider.protocol === "ollama" && shouldUseOllamaPoolForBaseUrl(this.baseUrl);
676695
- if (isOllama)
676778
+ const isOllamaTransport = this._provider.protocol === "ollama";
676779
+ const useOllamaPool = isOllamaTransport && shouldUseOllamaPoolForBaseUrl(this.baseUrl);
676780
+ if (isOllamaTransport)
676696
676781
  body["think"] = effectiveThink;
676697
676782
  if (responseFormat !== void 0) {
676698
676783
  body["response_format"] = responseFormat;
676699
676784
  }
676700
676785
  const reqNumCtx = request.numCtx ?? request.num_ctx;
676701
- if (isOllama && Number.isFinite(reqNumCtx) && (reqNumCtx ?? 0) > 0) {
676786
+ if (isOllamaTransport && Number.isFinite(reqNumCtx) && (reqNumCtx ?? 0) > 0) {
676702
676787
  const opts = body["options"] ?? {};
676703
676788
  opts["num_ctx"] = reqNumCtx;
676704
676789
  body["options"] = opts;
676705
676790
  body["num_ctx"] = reqNumCtx;
676706
676791
  }
676707
- let poolSlot = isOllama ? await getOllamaPool({ baseInstanceUrl: this.baseUrl }).acquire({
676792
+ let poolSlot = useOllamaPool ? await getOllamaPool({ baseInstanceUrl: this.baseUrl }).acquire({
676708
676793
  model: this.model,
676709
676794
  queuePolicy: request.poolQueuePolicy,
676710
676795
  queueTimeoutMs: request.poolQueueTimeoutMs
@@ -676736,7 +676821,7 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
676736
676821
  }
676737
676822
  }
676738
676823
  try {
676739
- if (isOllama && request.preferNativeOllamaChat === true) {
676824
+ if (isOllamaTransport && request.preferNativeOllamaChat === true) {
676740
676825
  const nativeOptions = {
676741
676826
  temperature: request.temperature,
676742
676827
  num_predict: effectiveMaxTokens
@@ -677255,6 +677340,7 @@ var init_nexusBackend = __esm({
677255
677340
  "packages/orchestrator/dist/nexusBackend.js"() {
677256
677341
  "use strict";
677257
677342
  init_textSanitize();
677343
+ init_ste_communication_policy();
677258
677344
  NexusAgenticBackend = class _NexusAgenticBackend {
677259
677345
  sendFn;
677260
677346
  model;
@@ -677288,7 +677374,7 @@ var init_nexusBackend = __esm({
677288
677374
  return messages2.map((message2) => typeof message2.content === "string" ? { ...message2, content: stripNoThinkPromptDirectives(message2.content) } : message2);
677289
677375
  }
677290
677376
  requestMessages(request, effectiveThink) {
677291
- return this.noThinkMessages(request.messages);
677377
+ return this.noThinkMessages(ensureSteCommunicationPolicy(request.messages));
677292
677378
  }
677293
677379
  applyOptionalRequestFields(daemonArgs, request) {
677294
677380
  const responseFormat = request.responseFormat ?? request.response_format;
@@ -684526,6 +684612,8 @@ __export(dist_exports3, {
684526
684612
  RequestFingerprintCache: () => RequestFingerprintCache,
684527
684613
  RetryController: () => RetryController,
684528
684614
  STALENESS_THRESHOLD_DEFAULTS: () => STALENESS_THRESHOLD_DEFAULTS,
684615
+ STE_COMMUNICATION_POLICY: () => STE_COMMUNICATION_POLICY,
684616
+ STE_POLICY_MARKER: () => STE_POLICY_MARKER,
684529
684617
  ScoutRunner: () => ScoutRunner,
684530
684618
  SessionMetrics: () => SessionMetrics,
684531
684619
  StreamingToolExecutor: () => StreamingToolExecutor,
@@ -684645,6 +684733,7 @@ __export(dist_exports3, {
684645
684733
  discoverPlugins: () => discoverPlugins3,
684646
684734
  discoverSystemOllamaModelStore: () => discoverSystemOllamaModelStore,
684647
684735
  effectiveContextWindow: () => effectiveContextWindow,
684736
+ ensureSteCommunicationPolicy: () => ensureSteCommunicationPolicy,
684648
684737
  estimateConservativeRequestInputTokens: () => estimateConservativeRequestInputTokens,
684649
684738
  evaluateDeliveryCoverage: () => evaluateDeliveryCoverage,
684650
684739
  evaluateExecutorStep: () => evaluateExecutorStep,
@@ -684884,6 +684973,7 @@ var init_dist8 = __esm({
684884
684973
  init_mergeRunner();
684885
684974
  init_retryController();
684886
684975
  init_agenticRunner();
684976
+ init_ste_communication_policy();
684887
684977
  init_request_budget();
684888
684978
  init_memory_compiler();
684889
684979
  init_memory_compiler_replay();
@@ -714982,7 +715072,7 @@ var init_task_templates = __esm({
714982
715072
  "git_info",
714983
715073
  "codebase_map"
714984
715074
  ],
714985
- outputGuidance: "Deliver working code with tests passing. Summarize what was changed and why."
715075
+ outputGuidance: "Deliver working code. Confirm that the applicable tests pass. Identify each change and its reason."
714986
715076
  },
714987
715077
  document: {
714988
715078
  type: "document",
@@ -714997,7 +715087,7 @@ var init_task_templates = __esm({
714997
715087
  "create_structured_file",
714998
715088
  "memory_read"
714999
715089
  ],
715000
- outputGuidance: "Deliver a complete, well-structured document. Specify the output format (Markdown, PDF, etc.)."
715090
+ outputGuidance: "Deliver a complete and well-structured document. Identify the output format."
715001
715091
  },
715002
715092
  analysis: {
715003
715093
  type: "analysis",
@@ -715016,7 +715106,7 @@ var init_task_templates = __esm({
715016
715106
  "create_structured_file",
715017
715107
  "shell"
715018
715108
  ],
715019
- outputGuidance: "Deliver a structured analysis with findings, data tables, and actionable recommendations."
715109
+ outputGuidance: "Deliver a structured analysis. Include findings, necessary data tables, and specific recommendations."
715020
715110
  },
715021
715111
  plan: {
715022
715112
  type: "plan",
@@ -715033,12 +715123,12 @@ var init_task_templates = __esm({
715033
715123
  "file_write",
715034
715124
  "git_info"
715035
715125
  ],
715036
- outputGuidance: "Deliver a structured plan with phases, milestones, risks, and action items."
715126
+ outputGuidance: "Deliver a structured plan. Include phases, milestones, risks, and specific actions."
715037
715127
  },
715038
715128
  general: {
715039
715129
  type: "general",
715040
715130
  label: "General Task",
715041
- description: "Tasks that don't clearly fit another category.",
715131
+ description: "Tasks that do not clearly fit another category.",
715042
715132
  systemPromptAddition: loadPrompt2("general.md"),
715043
715133
  recommendedTools: [
715044
715134
  "file_read",
@@ -715048,7 +715138,7 @@ var init_task_templates = __esm({
715048
715138
  "web_search",
715049
715139
  "memory_read"
715050
715140
  ],
715051
- outputGuidance: "Deliver complete results with a clear summary of what was done."
715141
+ outputGuidance: "Deliver complete results. Give a clear summary of the completed work."
715052
715142
  }
715053
715143
  };
715054
715144
  }
@@ -720885,6 +720975,7 @@ var init_status_bar = __esm({
720885
720975
  row: screenRow,
720886
720976
  lineText: hit.lineText
720887
720977
  })) {
720978
+ this.invalidateRowCountCache();
720888
720979
  if (this.active && !isOverlayActive() && !this._suspendContentLayer) {
720889
720980
  this.repaintContent();
720890
720981
  }
@@ -759451,46 +759542,55 @@ function buildRealtimeSystemPrompt(opts) {
759451
759542
  const voiceLimit = opts.maxVoiceChars ?? DEFAULT_VOICE_CHARS;
759452
759543
  const maxReplyWords = clampInt2(opts.maxReplyWords, DEFAULT_REALTIME_MAX_REPLY_WORDS, 8, 80);
759453
759544
  const sections = [
759545
+ STE_COMMUNICATION_POLICY,
759454
759546
  "[Omnius realtime conversation mode]",
759455
759547
  [
759456
- "Purpose: short, natural, back-and-forth spoken dialogue for ASR/TTS pipelines.",
759548
+ "Purpose: short and direct spoken dialog for ASR and TTS systems.",
759457
759549
  `Surface: ${surface}`,
759458
759550
  opts.sessionId ? `Session: ${opts.sessionId}` : "",
759459
759551
  `Workspace: ${repoRoot}`
759460
759552
  ].filter(Boolean).join("\n"),
759461
759553
  [
759462
759554
  "Interaction basis:",
759463
- "- Use dialogue alignment (Pickering/Garrod), grounding and repair, adjacency-pair response, and turn-taking timing principles (Wilson oscillator timing).",
759464
- "- Treat the latest user utterance as the live turn; if it interrupts or corrects earlier context, follow the latest turn.",
759465
- "- Listen for human cues in the provided words and conversation state; do not run local keyword classifiers."
759555
+ "- Use dialog alignment, grounding, repair, adjacent responses, and turn timing.",
759556
+ "- Treat the latest user statement as the current turn.",
759557
+ "- If that statement corrects earlier context, use the correction.",
759558
+ "- Use the supplied words and conversation state.",
759559
+ "- Do not use a local keyword classifier."
759466
759560
  ].join("\n"),
759467
759561
  [
759468
759562
  "Phone reply contract:",
759469
759563
  `- Produce one natural spoken turn, normally ${maxReplyWords} words or fewer.`,
759470
- "- Use one sentence when possible; two short sentences only when repair or confirmation needs it.",
759471
- "- Lead with the answer. Do not preface with status, analysis, summaries, or implementation narration.",
759472
- "- When the caller requests an exact literal response, emit only that literal text: no preface, suffix, punctuation, or commentary.",
759473
- "- No markdown, bullets, tables, headings, citations, inline code, code blocks, JSON, or labels like 'Assistant:'.",
759474
- "- Sound like a person on a live call: brief acknowledgment, direct answer, one focused follow-up only if needed.",
759475
- "- If the ASR text is garbled or underspecified, ask a single compact repair question.",
759476
- "- Do not invent app modes, method names, settings, or implementation details when the caller has not supplied them.",
759477
- "- Do not mention ASR, TTS, prompts, realtime mode, hidden reasoning, tools, or policy unless the caller explicitly asks.",
759478
- "- If a request needs work outside this text-only exchange, say the next handoff in one short sentence."
759564
+ "- Use one sentence when possible.",
759565
+ "- Use two short sentences only for repair or confirmation.",
759566
+ "- Start with the answer.",
759567
+ "- Do not add status, analysis, summaries, or implementation details before the answer.",
759568
+ "- If the caller requests exact text, return only that text.",
759569
+ "- Do not add punctuation or commentary to requested exact text.",
759570
+ "- Do not use document formatting.",
759571
+ "- Use a brief acknowledgment and a direct answer.",
759572
+ "- Ask one focused question only when necessary.",
759573
+ "- If the ASR text is unclear, ask one short repair question.",
759574
+ "- Do not invent modes, names, settings, or implementation details.",
759575
+ "- Do not mention system operation unless the caller requests that information.",
759576
+ "- If the request needs other work, identify the next transfer in one short sentence."
759479
759577
  ].join("\n"),
759480
759578
  soul ? `Project SOUL.md (${basename35(soul.path)}), compacted for realtime:
759481
759579
  ${blockText2(soul.content, soulLimit)}` : [
759482
759580
  "No project SOUL.md found. Default soul:",
759483
- "- pragmatic, direct, concrete",
759484
- "- low-fluff, human-paced, evidence-aware",
759485
- "- truthful about uncertainty without overexplaining"
759581
+ "- Use practical, direct, and concrete language.",
759582
+ "- Use a natural speaking rate.",
759583
+ "- Base each claim on evidence.",
759584
+ "- State uncertainty without unnecessary explanation."
759486
759585
  ].join("\n"),
759487
759586
  voice ? `Project voice profile (${basename35(voice.path)}), compacted for realtime:
759488
759587
  ${blockText2(voice.content, voiceLimit)}` : [
759489
759588
  "Default realtime voice:",
759490
- "- conversational, brief, and proportional",
759491
- "- phone-call natural: contractions, plain words, no written-document structure",
759492
- "- contractions are fine when natural",
759493
- "- no list formatting unless the user asks for a list"
759589
+ "- Use brief and proportional conversation.",
759590
+ "- Use plain words.",
759591
+ "- Do not use contractions.",
759592
+ "- Do not use document structure.",
759593
+ "- Do not use a list unless the user requests a list."
759494
759594
  ].join("\n")
759495
759595
  ];
759496
759596
  return sections.join("\n\n");
@@ -759565,7 +759665,7 @@ function wordParts(text3) {
759565
759665
  function finalizeRealtimeReply(text3, opts = {}) {
759566
759666
  const maxWords = clampInt2(opts.maxReplyWords, DEFAULT_REALTIME_MAX_REPLY_WORDS, 8, 80);
759567
759667
  let clean7 = stripHiddenThinking(String(text3 ?? "")).replace(/```[\s\S]*?```/g, "").split("\n").map((line) => line.replace(/^\s*(?:[-*]+|\d+[.)])\s+/, "").trim()).filter(Boolean).join(" ").replace(/^(?:assistant|omnius|agent)\s*:\s*/i, "").replace(/`([^`]+)`/g, "$1").replace(/\s+/g, " ").trim();
759568
- if (!clean7) return "I didn't catch that. Can you say it again?";
759668
+ if (!clean7) return "I did not understand that. Please say it again.";
759569
759669
  const sentences = clean7.match(/[^.!?]+[.!?]+(?=\s|$)|[^.!?]+$/g) ?? [clean7];
759570
759670
  const selected = [];
759571
759671
  let words = 0;
@@ -759590,6 +759690,7 @@ function finalizeRealtimeReply(text3, opts = {}) {
759590
759690
  var DEFAULT_REALTIME_HISTORY_MESSAGES, DEFAULT_REALTIME_MAX_TOKENS, DEFAULT_REALTIME_MAX_REPLY_WORDS, DEFAULT_SOUL_CHARS, DEFAULT_VOICE_CHARS;
759591
759691
  var init_realtime = __esm({
759592
759692
  "packages/cli/src/realtime.ts"() {
759693
+ init_dist8();
759593
759694
  DEFAULT_REALTIME_HISTORY_MESSAGES = 8;
759594
759695
  DEFAULT_REALTIME_MAX_TOKENS = 120;
759595
759696
  DEFAULT_REALTIME_MAX_REPLY_WORDS = 36;
@@ -759791,9 +759892,9 @@ function loadPersistedSessions() {
759791
759892
  return report2;
759792
759893
  }
759793
759894
  function buildSystemPrompt(cwd4) {
759794
- const parts = [];
759895
+ const parts = [STE_COMMUNICATION_POLICY];
759795
759896
  parts.push(
759796
- "You are Open Agent (Omnius), an AI coding assistant running locally via Ollama. You have access to the user's workspace and can discuss code, files, and projects. Be helpful, concise, and technically precise. When asked about files or code, describe what you know from the conversation context."
759897
+ "You are Open Agent, also named Omnius. You are an AI coding assistant. You can use the user's workspace. Give concise and precise technical information. For a file or code question, use facts from the conversation context."
759797
759898
  );
759798
759899
  parts.push(
759799
759900
  `\\nEnvironment: ${process.platform}, Node ${process.version}, CWD: ${cwd4}`
@@ -760185,6 +760286,7 @@ var sessions2, inFlight, SESSION_TTL_MS, INFERENCE_ROLES, PARTIAL_TAIL_BUDGET;
760185
760286
  var init_chat_session = __esm({
760186
760287
  "packages/cli/src/api/chat-session.ts"() {
760187
760288
  init_secret_redactor();
760289
+ init_dist8();
760188
760290
  init_session_quality();
760189
760291
  sessions2 = /* @__PURE__ */ new Map();
760190
760292
  inFlight = /* @__PURE__ */ new Map();
@@ -769026,6 +769128,7 @@ function createInferenceLiveBlockState(label = "Model inference", detail = "", o
769026
769128
  // This block exists specifically to expose the generated reasoning stream.
769027
769129
  // Callers may still hide only that section with Ctrl+O.
769028
769130
  showThinking: options2.showThinking ?? true,
769131
+ readMode: false,
769029
769132
  version: 0
769030
769133
  };
769031
769134
  }
@@ -769050,6 +769153,11 @@ function setInferenceThinkingVisible(state5, visible) {
769050
769153
  state5.showThinking = visible;
769051
769154
  state5.version++;
769052
769155
  }
769156
+ function setInferenceReadMode(state5, enabled2) {
769157
+ if (state5.readMode === enabled2) return;
769158
+ state5.readMode = enabled2;
769159
+ state5.version++;
769160
+ }
769053
769161
  function finishInferenceLiveBlock(state5, success, detail, now2 = Date.now()) {
769054
769162
  if (state5.status !== "running") return;
769055
769163
  if (detail) appendInferenceHandling(state5, detail);
@@ -769061,7 +769169,7 @@ function finishInferenceLiveBlock(state5, success, detail, now2 = Date.now()) {
769061
769169
  function inferenceLiveBlockFingerprint(state5, now2 = Date.now()) {
769062
769170
  const end = state5.finishedAt ?? now2;
769063
769171
  const sec = Math.floor(Math.max(0, end - state5.startedAt) / 1e3);
769064
- return `${state5.status}|${state5.phase}|${state5.version}|${state5.showThinking ? 1 : 0}|${sec}`;
769172
+ return `${state5.status}|${state5.phase}|${state5.version}|${state5.showThinking ? 1 : 0}|${state5.readMode ? 1 : 0}|${sec}`;
769065
769173
  }
769066
769174
  function buildInferenceLiveBlockLines(state5, width, now2 = Date.now()) {
769067
769175
  const w = Math.max(36, width);
@@ -769099,8 +769207,8 @@ function buildInferenceLiveBlockLines(state5, width, now2 = Date.now()) {
769099
769207
  if (visualRows.length === 0) {
769100
769208
  visualRows.push("handling › waiting for first generated token");
769101
769209
  }
769102
- const hidden = Math.max(0, visualRows.length - state5.maxRows);
769103
- const visibleRows = visualRows.slice(-state5.maxRows);
769210
+ const hidden = state5.readMode ? 0 : Math.max(0, visualRows.length - state5.maxRows);
769211
+ const visibleRows = state5.readMode ? visualRows : visualRows.slice(-state5.maxRows);
769104
769212
  const rows = [top];
769105
769213
  if (hidden > 0) {
769106
769214
  rows.push(
@@ -769111,6 +769219,8 @@ function buildInferenceLiveBlockLines(state5, width, now2 = Date.now()) {
769111
769219
  );
769112
769220
  }
769113
769221
  for (const row2 of visibleRows) rows.push(contentRow2(row2, inner));
769222
+ const readModeLabel = state5.readMode ? INFERENCE_EXIT_READ_MODE_LABEL : hidden > 0 ? `${INFERENCE_READ_MODE_LABEL} · show ${hidden} earlier row${hidden === 1 ? "" : "s"}` : INFERENCE_READ_MODE_LABEL;
769223
+ rows.push(contentRow2(`↳ ${readModeLabel}`, inner));
769114
769224
  rows.push(bottom);
769115
769225
  return rows;
769116
769226
  }
@@ -769170,9 +769280,11 @@ function formatElapsed2(ms) {
769170
769280
  const minutes = Math.floor(seconds / 60);
769171
769281
  return `${minutes}m${String(seconds % 60).padStart(2, "0")}s`;
769172
769282
  }
769173
- var ANSI_OR_OSC_RE, InferenceLiveBlockController;
769283
+ var INFERENCE_READ_MODE_LABEL, INFERENCE_EXIT_READ_MODE_LABEL, ANSI_OR_OSC_RE, InferenceLiveBlockController;
769174
769284
  var init_inference_live_block = __esm({
769175
769285
  "packages/cli/src/tui/inference-live-block.ts"() {
769286
+ INFERENCE_READ_MODE_LABEL = "read mode";
769287
+ INFERENCE_EXIT_READ_MODE_LABEL = "exit read mode";
769176
769288
  ANSI_OR_OSC_RE = /\x1B\[[0-?]*[ -/]*[@-~]|\x1B\].*?(?:\x07|\x1B\\)/g;
769177
769289
  InferenceLiveBlockController = class {
769178
769290
  constructor(host2) {
@@ -769195,6 +769307,15 @@ var init_inference_live_block = __esm({
769195
769307
  id2,
769196
769308
  (width) => buildInferenceLiveBlockLines(state5, width)
769197
769309
  );
769310
+ this.host.registerDynamicBlockClickHandler?.(id2, ({ lineText }) => {
769311
+ const plain = lineText.replace(ANSI_OR_OSC_RE, "").toLowerCase();
769312
+ if (!plain.includes(`↳ ${INFERENCE_READ_MODE_LABEL}`) && !plain.includes(`↳ ${INFERENCE_EXIT_READ_MODE_LABEL}`)) {
769313
+ return false;
769314
+ }
769315
+ setInferenceReadMode(state5, !state5.readMode);
769316
+ this.host.refreshDynamicBlocks?.();
769317
+ return true;
769318
+ });
769198
769319
  this.host.appendDynamicBlock(id2);
769199
769320
  this.schedule(key, block);
769200
769321
  return state5;
@@ -783959,10 +784080,11 @@ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
783959
784080
  } else if (chunk.type === "finish") {
783960
784081
  finishReason = chunk.finishReason;
783961
784082
  } else if (chunk.type === "usage") {
784083
+ const chunkUsage = chunk.usage;
783962
784084
  usage = {
783963
- prompt_tokens: chunk.promptTokens,
783964
- completion_tokens: chunk.completionTokens,
783965
- total_tokens: chunk.totalTokens
784085
+ prompt_tokens: chunkUsage?.promptTokens,
784086
+ completion_tokens: chunkUsage?.completionTokens,
784087
+ total_tokens: chunkUsage?.totalTokens
783966
784088
  };
783967
784089
  }
783968
784090
  }
@@ -787845,6 +787967,7 @@ ${conversationStream}`
787845
787967
  });
787846
787968
  let accumulated = "";
787847
787969
  let streamError;
787970
+ let inferenceSucceeded = false;
787848
787971
  const sessionKey = this.sessionKeyForMessage(msg);
787849
787972
  const inferenceId = this.registerTelegramInference(
787850
787973
  "chat-fast-path",
@@ -787858,11 +787981,28 @@ ${conversationStream}`
787858
787981
  try {
787859
787982
  let lastTokenEmitMs = 0;
787860
787983
  for await (const chunk of stream) {
787984
+ if (chunk.type === "tool_call_delta") {
787985
+ const contractDelta = `${chunk.toolCallName ? `name=${chunk.toolCallName}
787986
+ arguments=` : ""}` + (chunk.toolCallArgs ?? "");
787987
+ if (contractDelta) {
787988
+ this.telegramInferenceBlocks?.append(
787989
+ inferenceId,
787990
+ "tool_args",
787991
+ getSecretRedactor().redactText(contractDelta)
787992
+ );
787993
+ }
787994
+ continue;
787995
+ }
787861
787996
  if (chunk.type !== "content") continue;
787862
787997
  const piece = chunk.content;
787863
787998
  if (!piece) continue;
787864
787999
  if (chunk.thinking) {
787865
788000
  this.bumpTelegramInferenceTokens(inferenceId, 0, 1);
788001
+ this.telegramInferenceBlocks?.append(
788002
+ inferenceId,
788003
+ "thinking",
788004
+ getSecretRedactor().redactText(piece)
788005
+ );
787866
788006
  if (this.telegramThinkingVisible) {
787867
788007
  const preview = piece.slice(0, 120);
787868
788008
  this.tuiWrite(
@@ -787874,6 +788014,11 @@ ${conversationStream}`
787874
788014
  }
787875
788015
  } else {
787876
788016
  this.bumpTelegramInferenceTokens(inferenceId, 1, 0);
788017
+ this.telegramInferenceBlocks?.append(
788018
+ inferenceId,
788019
+ "content",
788020
+ getSecretRedactor().redactText(piece)
788021
+ );
787877
788022
  accumulated += piece;
787878
788023
  const now2 = Date.now();
787879
788024
  if (now2 - lastTokenEmitMs > 120) {
@@ -787893,6 +788038,10 @@ ${conversationStream}`
787893
788038
  } catch (err) {
787894
788039
  streamError = err;
787895
788040
  accumulated = "";
788041
+ this.telegramInferenceBlocks?.handling(
788042
+ inferenceId,
788043
+ `stream failed; falling back to unary response: ${err instanceof Error ? err.message : String(err)}`
788044
+ );
787896
788045
  }
787897
788046
  }
787898
788047
  if (!accumulated.trim()) {
@@ -787914,7 +788063,17 @@ ${conversationStream}`
787914
788063
  const fullExtracted = extractPartialTelegramReplyJson(accumulated);
787915
788064
  if (fullExtracted) await onToken(fullExtracted);
787916
788065
  }
788066
+ this.telegramInferenceBlocks?.handling(
788067
+ inferenceId,
788068
+ "response received; handing returned contract to parser"
788069
+ );
788070
+ inferenceSucceeded = true;
787917
788071
  } finally {
788072
+ this.telegramInferenceBlocks?.finish(
788073
+ inferenceId,
788074
+ inferenceSucceeded,
788075
+ inferenceSucceeded ? void 0 : "inference ended before a response contract was available"
788076
+ );
787918
788077
  this.deregisterTelegramInference(inferenceId);
787919
788078
  }
787920
788079
  const extracted = extractFinalTelegramReplyJson(accumulated);
@@ -799633,7 +799792,8 @@ ${entry.fullContent}`
799633
799792
  const inferenceBlocks = statusBar?.isActive && !isNeovimActive() ? new InferenceLiveBlockController({
799634
799793
  registerDynamicBlock: (id2, render2) => statusBar.registerDynamicBlock(id2, render2),
799635
799794
  appendDynamicBlock: (id2) => statusBar.appendDynamicBlock(id2),
799636
- refreshDynamicBlocks: () => statusBar.refreshDynamicBlocks()
799795
+ refreshDynamicBlocks: () => statusBar.refreshDynamicBlocks(),
799796
+ registerDynamicBlockClickHandler: (id2, handler) => statusBar.registerDynamicBlockClickHandler(id2, handler)
799637
799797
  }) : null;
799638
799798
  _mainInferenceBlocksRef = inferenceBlocks;
799639
799799
  const mainInferenceBlockKey = "main";
@@ -804194,7 +804354,8 @@ Respond concisely and safely. Remember: you are talking to the general public.`;
804194
804354
  telegramBridge.setLiveInferenceHost({
804195
804355
  registerDynamicBlock: (id2, render2) => statusBar.registerDynamicBlock(id2, render2),
804196
804356
  appendDynamicBlock: (id2) => statusBar.appendDynamicBlock(id2),
804197
- refreshDynamicBlocks: () => statusBar.refreshDynamicBlocks()
804357
+ refreshDynamicBlocks: () => statusBar.refreshDynamicBlocks(),
804358
+ registerDynamicBlockClickHandler: (id2, handler) => statusBar.registerDynamicBlockClickHandler(id2, handler)
804198
804359
  });
804199
804360
  telegramBridge.setSubAgentViewCallbacks({
804200
804361
  onRegister: (id2, label, taskSummary) => {