@mmnto/cli 1.101.1 → 1.102.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/dist/commands/doctor.d.ts +36 -6
  2. package/dist/commands/doctor.d.ts.map +1 -1
  3. package/dist/commands/doctor.js +27 -0
  4. package/dist/commands/doctor.js.map +1 -1
  5. package/dist/commands/doctor.test.js +46 -1
  6. package/dist/commands/doctor.test.js.map +1 -1
  7. package/dist/commands/ecl-gc.d.ts.map +1 -1
  8. package/dist/commands/ecl-gc.js +16 -4
  9. package/dist/commands/ecl-gc.js.map +1 -1
  10. package/dist/commands/eject.d.ts +37 -2
  11. package/dist/commands/eject.d.ts.map +1 -1
  12. package/dist/commands/eject.js +79 -9
  13. package/dist/commands/eject.js.map +1 -1
  14. package/dist/commands/eject.test.js +169 -7
  15. package/dist/commands/eject.test.js.map +1 -1
  16. package/dist/commands/gate-install.test.js +22 -1
  17. package/dist/commands/gate-install.test.js.map +1 -1
  18. package/dist/commands/init-detect.js +1 -1
  19. package/dist/commands/init-detect.js.map +1 -1
  20. package/dist/commands/init-detect.test.js +2 -1
  21. package/dist/commands/init-detect.test.js.map +1 -1
  22. package/dist/commands/init-templates.js +8 -8
  23. package/dist/commands/install-hooks.d.ts +16 -0
  24. package/dist/commands/install-hooks.d.ts.map +1 -1
  25. package/dist/commands/install-hooks.js +5 -1
  26. package/dist/commands/install-hooks.js.map +1 -1
  27. package/dist/commands/mail-cli-wiring.test.d.ts +15 -0
  28. package/dist/commands/mail-cli-wiring.test.d.ts.map +1 -0
  29. package/dist/commands/mail-cli-wiring.test.js +106 -0
  30. package/dist/commands/mail-cli-wiring.test.js.map +1 -0
  31. package/dist/commands/mail.d.ts +80 -0
  32. package/dist/commands/mail.d.ts.map +1 -1
  33. package/dist/commands/mail.js +241 -9
  34. package/dist/commands/mail.js.map +1 -1
  35. package/dist/commands/mail.test.js +359 -4
  36. package/dist/commands/mail.test.js.map +1 -1
  37. package/dist/commands/review-fan.d.ts.map +1 -1
  38. package/dist/commands/review-fan.js +32 -8
  39. package/dist/commands/review-fan.js.map +1 -1
  40. package/dist/commands/review-fan.test.js +153 -7
  41. package/dist/commands/review-fan.test.js.map +1 -1
  42. package/dist/commands/shield-generated.d.ts +145 -0
  43. package/dist/commands/shield-generated.d.ts.map +1 -0
  44. package/dist/commands/shield-generated.js +266 -0
  45. package/dist/commands/shield-generated.js.map +1 -0
  46. package/dist/commands/shield-generated.test.d.ts +2 -0
  47. package/dist/commands/shield-generated.test.d.ts.map +1 -0
  48. package/dist/commands/shield-generated.test.js +323 -0
  49. package/dist/commands/shield-generated.test.js.map +1 -0
  50. package/dist/commands/shield.d.ts +27 -5
  51. package/dist/commands/shield.d.ts.map +1 -1
  52. package/dist/commands/shield.js +116 -41
  53. package/dist/commands/shield.js.map +1 -1
  54. package/dist/commands/shield.test.js +92 -2
  55. package/dist/commands/shield.test.js.map +1 -1
  56. package/dist/commands/status.d.ts.map +1 -1
  57. package/dist/commands/status.js +26 -2
  58. package/dist/commands/status.js.map +1 -1
  59. package/dist/commands/status.test.js +55 -1
  60. package/dist/commands/status.test.js.map +1 -1
  61. package/dist/hook/schema.d.ts +2 -2
  62. package/dist/index.js +28 -7
  63. package/dist/index.js.map +1 -1
  64. package/dist/orchestrators/orchestrator.d.ts +86 -1
  65. package/dist/orchestrators/orchestrator.d.ts.map +1 -1
  66. package/dist/orchestrators/orchestrator.js +277 -9
  67. package/dist/orchestrators/orchestrator.js.map +1 -1
  68. package/dist/orchestrators/orchestrator.test.js +240 -4
  69. package/dist/orchestrators/orchestrator.test.js.map +1 -1
  70. package/dist/orchestrators/shell-orchestrator.d.ts +12 -1
  71. package/dist/orchestrators/shell-orchestrator.d.ts.map +1 -1
  72. package/dist/orchestrators/shell-orchestrator.js +256 -49
  73. package/dist/orchestrators/shell-orchestrator.js.map +1 -1
  74. package/dist/orchestrators/shell-orchestrator.test.js +252 -4
  75. package/dist/orchestrators/shell-orchestrator.test.js.map +1 -1
  76. package/dist/utils.d.ts +19 -1
  77. package/dist/utils.d.ts.map +1 -1
  78. package/dist/utils.js +394 -108
  79. package/dist/utils.js.map +1 -1
  80. package/dist/utils.test.js +406 -14
  81. package/dist/utils.test.js.map +1 -1
  82. package/package.json +2 -2
package/dist/utils.js CHANGED
@@ -3,8 +3,8 @@ import * as fs from 'node:fs';
3
3
  import * as os from 'node:os';
4
4
  import * as path from 'node:path';
5
5
  import dotenv from 'dotenv';
6
- import { ADMISSION_COMPLETION_ONLY, buildGroundingBundle, calculateDeterministicHash, CONFIG_FILES, maskSecrets, RUN_ARTIFACT_SCHEMA_VERSION, saveRunArtifact, TotemConfigError, TotemConfigSchema, TotemOrchestratorError, } from '@mmnto/totem';
7
- import { createOrchestrator, resolveOrchestrator } from './orchestrators/orchestrator.js';
6
+ import { ADMISSION_COMPLETION_ONLY, buildGroundingBundle, calculateDeterministicHash, CONFIG_FILES, INVOCATION_FAILURE_ARTIFACT_SCHEMA_VERSION, INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES, INVOKE_STREAM_EVIDENCE_LIMIT_BYTES, maskSecrets, RUN_ARTIFACT_SCHEMA_VERSION, SAFE_PROVIDER_CODE_RE, sanitizeForTerminal, saveInvocationFailureArtifact, saveRunArtifact, TotemConfigError, TotemConfigSchema, TotemOrchestratorError, } from '@mmnto/totem';
7
+ import { classifyInvokeFailure, createOrchestrator, OrchestratorInvokeError, resolveOrchestrator, toOrchestratorInvokeError, } from './orchestrators/orchestrator.js';
8
8
  import { bold, log } from './ui.js';
9
9
  // ─── Shared constants ────────────────────────────────────
10
10
  const TELEMETRY_FILE = 'telemetry.jsonl';
@@ -197,6 +197,11 @@ export { sanitize } from '@mmnto/totem';
197
197
  // ─── XML delimiting ─────────────────────────────────────
198
198
  // Re-export from core — unified XML escaping (#158)
199
199
  export { wrapUntrustedXml, wrapXml } from '@mmnto/totem';
200
+ // ─── Glob matching ──────────────────────────────────────
201
+ // Re-export from core — shared glob semantics so CLI command modules avoid a
202
+ // static top-level barrel import of '@mmnto/totem' (heavy-deps-at-startup rule,
203
+ // mmnto-ai/totem#2339). Used by the review generated-artifact classifier (#2398).
204
+ export { matchesGlob } from '@mmnto/totem';
200
205
  // ─── Context formatting ─────────────────────────────────
201
206
  const MAX_RESULT_CONTENT_LENGTH = 300;
202
207
  const CONDENSED_CONTENT_LENGTH = 80;
@@ -317,6 +322,231 @@ function buildResponseCacheHash(prompt, systemPrompt, qualifiedModel, contract)
317
322
  }
318
323
  return hash.digest('hex').slice(0, 16);
319
324
  }
325
+ function takeUtf8Prefix(text, limitBytes) {
326
+ let retained = '';
327
+ let bytes = 0;
328
+ for (const character of text) {
329
+ const characterBytes = Buffer.byteLength(character, 'utf-8');
330
+ if (bytes + characterBytes > limitBytes)
331
+ break;
332
+ retained += character;
333
+ bytes += characterBytes;
334
+ }
335
+ return retained;
336
+ }
337
+ function takeUtf8Tail(text, limitBytes) {
338
+ const characters = Array.from(text);
339
+ const retainedReversed = [];
340
+ let bytes = 0;
341
+ for (let index = characters.length - 1; index >= 0; index--) {
342
+ const character = characters[index];
343
+ const characterBytes = Buffer.byteLength(character, 'utf-8');
344
+ if (bytes + characterBytes > limitBytes)
345
+ break;
346
+ retainedReversed.push(character);
347
+ bytes += characterBytes;
348
+ }
349
+ return retainedReversed.reverse().join('');
350
+ }
351
+ /**
352
+ * Convert bounded raw runtime text into persisted evidence. Masking and
353
+ * terminal sanitization happen before a second UTF-8 byte bound because a
354
+ * replacement may grow or shrink the retained text. A masking exception
355
+ * fails closed: no raw bytes cross the artifact boundary.
356
+ */
357
+ export function persistRuntimeTextEvidence(runtime, customSecrets, limitBytes = INVOKE_STREAM_EVIDENCE_LIMIT_BYTES, masker = maskSecrets) {
358
+ let safeHead;
359
+ let safeTail;
360
+ try {
361
+ safeHead = sanitizeForTerminal(masker(sanitizeForTerminal(runtime.head), customSecrets));
362
+ safeTail =
363
+ runtime.tail === undefined
364
+ ? undefined
365
+ : sanitizeForTerminal(masker(sanitizeForTerminal(runtime.tail), customSecrets));
366
+ // totem-context: evidence masking fails closed by intentionally omitting all raw text and recording the typed omission marker below.
367
+ }
368
+ catch {
369
+ return {
370
+ encoding: 'utf-8',
371
+ head: '',
372
+ observedBytes: runtime.observedBytes,
373
+ retainedBytes: 0,
374
+ limitBytes,
375
+ truncated: false,
376
+ dlp: 'omitted-on-mask-failure',
377
+ };
378
+ }
379
+ const hasTail = safeTail !== undefined;
380
+ const headLimit = hasTail ? Math.floor(limitBytes / 2) : limitBytes;
381
+ const tailLimit = limitBytes - headLimit;
382
+ const head = takeUtf8Prefix(safeHead, headLimit);
383
+ const tail = safeTail === undefined ? undefined : takeUtf8Tail(safeTail, tailLimit);
384
+ const retainedBytes = Buffer.byteLength(head, 'utf-8') + Buffer.byteLength(tail ?? '', 'utf-8');
385
+ const postMaskTruncated = Buffer.byteLength(safeHead, 'utf-8') > headLimit ||
386
+ (safeTail !== undefined && Buffer.byteLength(safeTail, 'utf-8') > tailLimit);
387
+ return {
388
+ encoding: 'utf-8',
389
+ head,
390
+ ...(tail !== undefined ? { tail } : {}),
391
+ observedBytes: runtime.observedBytes,
392
+ retainedBytes,
393
+ limitBytes,
394
+ truncated: runtime.truncated || postMaskTruncated,
395
+ dlp: 'masked',
396
+ };
397
+ }
398
+ function persistRuntimeAttempts(attempts, customSecrets) {
399
+ return attempts.map((attempt, index) => {
400
+ const providerCode = persistProviderCode(attempt.providerCode, customSecrets);
401
+ return {
402
+ sequence: index + 1,
403
+ route: attempt.route,
404
+ provider: attempt.provider,
405
+ model: attempt.model,
406
+ status: attempt.status,
407
+ durationMs: attempt.durationMs,
408
+ ...(attempt.failureKind !== undefined ? { failureKind: attempt.failureKind } : {}),
409
+ ...(attempt.providerStatus !== undefined ? { providerStatus: attempt.providerStatus } : {}),
410
+ ...(providerCode !== undefined ? { providerCode } : {}),
411
+ ...(attempt.process !== undefined
412
+ ? {
413
+ process: {
414
+ exitCode: attempt.process.exitCode,
415
+ signal: attempt.process.signal,
416
+ timedOut: attempt.process.timedOut,
417
+ ...(attempt.process.timeoutMs !== undefined
418
+ ? { timeoutMs: attempt.process.timeoutMs }
419
+ : {}),
420
+ ...(attempt.process.stdout !== undefined
421
+ ? { stdout: persistRuntimeTextEvidence(attempt.process.stdout, customSecrets) }
422
+ : {}),
423
+ ...(attempt.process.stderr !== undefined
424
+ ? { stderr: persistRuntimeTextEvidence(attempt.process.stderr, customSecrets) }
425
+ : {}),
426
+ },
427
+ }
428
+ : {}),
429
+ };
430
+ });
431
+ }
432
+ function persistProviderCode(providerCode, customSecrets) {
433
+ if (providerCode === undefined || !SAFE_PROVIDER_CODE_RE.test(providerCode))
434
+ return undefined;
435
+ try {
436
+ return maskSecrets(providerCode, customSecrets) === providerCode ? providerCode : undefined;
437
+ // totem-context: provider codes are optional diagnostics; a masking failure intentionally omits the code rather than persisting an unsafe token.
438
+ }
439
+ catch {
440
+ return undefined;
441
+ }
442
+ }
443
+ /**
444
+ * Pre-bound provider-controlled terminal prose before DLP. Secret masking can
445
+ * expand retained text, so `persistRuntimeTextEvidence` applies the same cap a
446
+ * second time after masking; this first bound prevents unbounded regex work.
447
+ */
448
+ export function runtimeMessageEvidence(message) {
449
+ const observedBytes = Buffer.byteLength(message, 'utf-8');
450
+ if (observedBytes <= INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES) {
451
+ return {
452
+ encoding: 'utf-8',
453
+ head: message,
454
+ observedBytes,
455
+ retainedBytes: observedBytes,
456
+ limitBytes: INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES,
457
+ truncated: false,
458
+ };
459
+ }
460
+ const headLimit = Math.floor(INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES / 2);
461
+ const tailLimit = INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES - headLimit;
462
+ const head = takeUtf8Prefix(message, headLimit);
463
+ const tail = takeUtf8Tail(message, tailLimit);
464
+ const retainedBytes = Buffer.byteLength(head, 'utf-8') + Buffer.byteLength(tail, 'utf-8');
465
+ return {
466
+ encoding: 'utf-8',
467
+ head,
468
+ tail,
469
+ observedBytes,
470
+ retainedBytes,
471
+ limitBytes: INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES,
472
+ truncated: true,
473
+ };
474
+ }
475
+ function buildArtifactSharedFields(args) {
476
+ const inputBundle = {
477
+ maskedPrompt: args.safePrompt,
478
+ ...(args.safeSystemPrompt !== undefined && args.safeSystemPrompt.length > 0
479
+ ? { maskedSystemPrompt: args.safeSystemPrompt }
480
+ : {}),
481
+ ...(args.artifact.diffScope !== undefined ? { diffScope: args.artifact.diffScope } : {}),
482
+ ...(args.artifact.specContract !== undefined
483
+ ? { specContract: args.artifact.specContract }
484
+ : {}),
485
+ };
486
+ return {
487
+ inputBundle,
488
+ inputHash: calculateDeterministicHash(inputBundle),
489
+ grounding: {
490
+ hash: args.artifact.groundingHash,
491
+ provenanceSummary: args.artifact.provenanceSummary,
492
+ ...(args.groundingBundle !== undefined ? { bundle: args.groundingBundle } : {}),
493
+ },
494
+ ...(args.outputContract !== undefined ||
495
+ args.contextPolicy !== undefined ||
496
+ args.runMetadata !== undefined
497
+ ? {
498
+ admission: {
499
+ ...(args.outputContract !== undefined ? { outputContract: args.outputContract } : {}),
500
+ ...(args.contextPolicy !== undefined ? { contextPolicy: args.contextPolicy } : {}),
501
+ ...(args.runMetadata !== undefined ? { runMetadata: args.runMetadata } : {}),
502
+ },
503
+ }
504
+ : {}),
505
+ };
506
+ }
507
+ function buildArtifactBackend(args) {
508
+ return {
509
+ provider: args.provider,
510
+ model: args.model,
511
+ qualifiedModel: args.qualifiedModel,
512
+ admissionClass: args.admissionClass,
513
+ taskProfile: args.taskProfile,
514
+ ...(args.temperature !== undefined ? { temperature: args.temperature } : {}),
515
+ };
516
+ }
517
+ function runtimeAttemptForError(args) {
518
+ return {
519
+ sequence: 1,
520
+ route: args.route,
521
+ provider: args.provider,
522
+ model: args.model,
523
+ status: 'failed',
524
+ durationMs: 0,
525
+ failureKind: classifyInvokeFailure(args.err),
526
+ ...(args.err instanceof Error &&
527
+ 'status' in args.err &&
528
+ typeof args.err.status === 'number'
529
+ ? { providerStatus: args.err.status }
530
+ : {}),
531
+ ...(args.err instanceof Error &&
532
+ 'code' in args.err &&
533
+ typeof args.err.code === 'string'
534
+ ? { providerCode: args.err.code }
535
+ : {}),
536
+ };
537
+ }
538
+ function runtimeAttemptsForError(err, provider, model, route) {
539
+ if (err instanceof OrchestratorInvokeError && err.attempts.length > 0) {
540
+ return err.attempts.map((attempt) => ({ ...attempt }));
541
+ }
542
+ return [runtimeAttemptForError({ err, provider, model, route })];
543
+ }
544
+ function resequenceRuntimeAttempts(attempts) {
545
+ return attempts.map((attempt, index) => ({ ...attempt, sequence: index + 1 }));
546
+ }
547
+ function quotaFallbackAttempts(attempts) {
548
+ return attempts.map((attempt) => ({ ...attempt, route: 'quota-model-fallback' }));
549
+ }
320
550
  /**
321
551
  * Assemble the grounding bundle for the spec/review retrieval shape
322
552
  * (mmnto-ai/totem#2101): every partition's items enter under their partition
@@ -497,7 +727,7 @@ export async function runOrchestrator(opts) {
497
727
  let resolved = resolveOrchestrator(rawModel, baseProvider, baseInvoke);
498
728
  let model = resolved.parsed.model;
499
729
  let qualifiedModel = resolved.qualifiedModel;
500
- let invoke = resolved.invoke;
730
+ const invoke = resolved.invoke;
501
731
  // ── Admission gate, primary path (mmnto-ai/totem#2102) ──
502
732
  // Decided per RESOLVED backend BEFORE the invoke (and before the response
503
733
  // cache: a denied class must not be served a replay either). No tokens are
@@ -593,78 +823,156 @@ export async function runOrchestrator(opts) {
593
823
  ...(opts.outputContract !== undefined ? { outputContract: opts.outputContract } : {}),
594
824
  ...(opts.runMetadata !== undefined ? { runMetadata: opts.runMetadata } : {}),
595
825
  };
826
+ const requestedBackend = buildArtifactBackend({
827
+ provider: resolved.parsed.provider,
828
+ model,
829
+ qualifiedModel,
830
+ admissionClass: requestedAdmissionClass,
831
+ taskProfile,
832
+ ...(opts.temperature !== undefined ? { temperature: opts.temperature } : {}),
833
+ });
596
834
  let result;
835
+ const primaryStartMs = Date.now();
597
836
  try {
598
- result = await invoke({
599
- prompt: safePrompt,
600
- ...(safeSystemPrompt !== undefined ? { systemPrompt: safeSystemPrompt } : {}),
601
- model,
602
- cwd,
603
- tag,
604
- totemDir: config.totemDir,
605
- temperature: opts.temperature,
606
- ...(enableContextCaching !== undefined ? { enableContextCaching } : {}),
607
- ...(cacheTTL !== undefined ? { cacheTTL } : {}),
608
- ...admissionTransport,
609
- });
610
- }
611
- catch (err) {
612
- if (err instanceof Error && err.name === 'QuotaError') {
837
+ try {
838
+ result = await invoke({
839
+ prompt: safePrompt,
840
+ ...(safeSystemPrompt !== undefined ? { systemPrompt: safeSystemPrompt } : {}),
841
+ model,
842
+ cwd,
843
+ tag,
844
+ totemDir: config.totemDir,
845
+ temperature: opts.temperature,
846
+ ...(enableContextCaching !== undefined ? { enableContextCaching } : {}),
847
+ ...(cacheTTL !== undefined ? { cacheTTL } : {}),
848
+ ...admissionTransport,
849
+ });
850
+ }
851
+ catch (err) {
852
+ const primaryErr = toOrchestratorInvokeError({
853
+ err,
854
+ provider: resolved.parsed.provider,
855
+ model,
856
+ route: 'sdk',
857
+ durationMs: Date.now() - primaryStartMs,
858
+ });
859
+ if (primaryErr.kind !== 'quota')
860
+ throw primaryErr;
861
+ const primaryAttempts = runtimeAttemptsForError(primaryErr, resolved.parsed.provider, model, 'sdk');
613
862
  const rawFallback = config.orchestrator.fallbackModel;
614
- if (rawFallback && rawModel !== rawFallback) {
615
- log.warn(tag, `Quota exhausted for ${rawModel}. Retrying with fallback model: ${bold(rawFallback)}...`);
616
- const fallbackResolved = resolveOrchestrator(rawFallback, baseProvider, baseInvoke);
617
- // ── Admission gate, fallback path (mmnto-ai/totem#2102) ──
618
- // `resolveOrchestrator` can route a provider-qualified fallbackModel
619
- // to a DIFFERENT provider, and a single config-level declaration
620
- // cannot honestly cover backends with different real capabilities.
621
- // Slice-3 rule, conservative and deterministic: an elevated class
622
- // admits the fallback only when it resolves to the SAME provider as
623
- // the primary cross-provider fails loud BEFORE the fallback invoke,
624
- // reporting the primary and admission errors together. Per-provider
625
- // capability declarations are the future relaxation.
626
- if (requestedAdmissionClass !== ADMISSION_COMPLETION_ONLY &&
627
- fallbackResolved.parsed.provider !== resolved.parsed.provider) {
628
- throw new TotemOrchestratorError(`Primary model '${rawModel}' failed and the quota fallback '${rawFallback}' was denied admission.\n\n` +
629
- `Primary error:\n${err.message}\n\n` +
630
- `Admission error:\nfallback resolves to provider '${fallbackResolved.parsed.provider}' (primary: '${resolved.parsed.provider}') while admission class '${requestedAdmissionClass}' is requested a cross-provider fallback is not admitted above '${ADMISSION_COMPLETION_ONLY}'.`, 'Use a same-provider fallbackModel, or drop the elevated backendAdmissionClass request.', err);
631
- }
632
- try {
633
- result = await fallbackResolved.invoke({
634
- prompt: safePrompt,
635
- ...(safeSystemPrompt !== undefined ? { systemPrompt: safeSystemPrompt } : {}),
863
+ if (!rawFallback || rawModel === rawFallback) {
864
+ if (err instanceof OrchestratorInvokeError)
865
+ throw primaryErr;
866
+ throw new OrchestratorInvokeError(`Quota exhausted for ${model}.`, 'quota', resequenceRuntimeAttempts(primaryAttempts), {
867
+ cause: err,
868
+ recoveryHint: 'Wait for quota to reset, configure orchestrator.fallbackModel, or select another model.',
869
+ });
870
+ }
871
+ log.warn(tag, `Quota exhausted for ${rawModel}. Retrying with fallback model: ${bold(rawFallback)}...`);
872
+ const fallbackResolved = resolveOrchestrator(rawFallback, baseProvider, baseInvoke);
873
+ // ── Admission gate, fallback path (mmnto-ai/totem#2102) ──
874
+ // `resolveOrchestrator` can route a provider-qualified fallbackModel
875
+ // to a DIFFERENT provider, and a single config-level declaration
876
+ // cannot honestly cover backends with different real capabilities.
877
+ // Slice-3 rule, conservative and deterministic: an elevated class
878
+ // admits the fallback only when it resolves to the SAME provider as
879
+ // the primary — cross-provider fails loud BEFORE the fallback invoke.
880
+ if (requestedAdmissionClass !== ADMISSION_COMPLETION_ONLY &&
881
+ fallbackResolved.parsed.provider !== resolved.parsed.provider) {
882
+ throw new OrchestratorInvokeError(`Primary model '${rawModel}' failed and the quota fallback '${rawFallback}' was denied admission.\n\n` +
883
+ `Primary error:\n${primaryErr.message}\n\n` +
884
+ `Admission error:\nfallback resolves to provider '${fallbackResolved.parsed.provider}' (primary: '${resolved.parsed.provider}') while admission class '${requestedAdmissionClass}' is requested — a cross-provider fallback is not admitted above '${ADMISSION_COMPLETION_ONLY}'.`, 'quota', resequenceRuntimeAttempts(primaryAttempts), {
885
+ cause: primaryErr,
886
+ recoveryHint: 'Use a same-provider fallbackModel, or drop the elevated backendAdmissionClass request.',
887
+ });
888
+ }
889
+ const fallbackStartMs = Date.now();
890
+ try {
891
+ const fallbackResult = await fallbackResolved.invoke({
892
+ prompt: safePrompt,
893
+ ...(safeSystemPrompt !== undefined ? { systemPrompt: safeSystemPrompt } : {}),
894
+ model: fallbackResolved.parsed.model,
895
+ cwd,
896
+ tag,
897
+ totemDir: config.totemDir,
898
+ temperature: opts.temperature,
899
+ ...(enableContextCaching !== undefined ? { enableContextCaching } : {}),
900
+ ...(cacheTTL !== undefined ? { cacheTTL } : {}),
901
+ ...admissionTransport,
902
+ });
903
+ const fallbackAttempts = quotaFallbackAttempts(fallbackResult.attempts ?? [
904
+ {
905
+ sequence: 1,
906
+ route: 'quota-model-fallback',
907
+ provider: fallbackResolved.parsed.provider,
636
908
  model: fallbackResolved.parsed.model,
637
- cwd,
638
- tag,
639
- totemDir: config.totemDir,
640
- temperature: opts.temperature,
641
- ...(enableContextCaching !== undefined ? { enableContextCaching } : {}),
642
- ...(cacheTTL !== undefined ? { cacheTTL } : {}),
643
- ...admissionTransport,
644
- });
645
- // Update model/invoke so telemetry and cache log the correct values
646
- model = fallbackResolved.parsed.model;
647
- qualifiedModel = fallbackResolved.qualifiedModel;
648
- resolved = fallbackResolved;
649
- invoke = fallbackResolved.invoke;
650
- }
651
- catch (fallbackErr) {
652
- const originalMsg = err.message;
653
- const fallbackMsg = fallbackErr instanceof Error ? fallbackErr.message : String(fallbackErr);
654
- throw new TotemOrchestratorError(`Primary model '${rawModel}' failed and fallback model '${rawFallback}' also failed.\n\n` +
655
- `Primary error:\n${originalMsg}\n\nFallback error:\n${fallbackMsg}`, 'Check API quotas and model availability, or try a different model with --model.', fallbackErr);
656
- }
909
+ status: 'succeeded',
910
+ durationMs: fallbackResult.durationMs,
911
+ },
912
+ ]);
913
+ result = {
914
+ ...fallbackResult,
915
+ attempts: resequenceRuntimeAttempts([...primaryAttempts, ...fallbackAttempts]),
916
+ };
917
+ // Update the resolved backend identity so telemetry, cache, and success
918
+ // artifacts log the backend that actually produced the semantic output.
919
+ model = fallbackResolved.parsed.model;
920
+ qualifiedModel = fallbackResolved.qualifiedModel;
921
+ resolved = fallbackResolved;
657
922
  }
658
- else {
659
- throw new TotemOrchestratorError(`Quota exhausted for ${model}.`, 'Quota resets on a rolling daily window. Options:\n' +
660
- ' - Switch to a flash model: totem <command> --model <name>\n' +
661
- ' - Inspect the prompt without calling the API: totem <command> --raw\n' +
662
- ' - Set a fallbackModel in totem.config.ts');
923
+ catch (fallbackErr) {
924
+ const normalizedFallbackErr = toOrchestratorInvokeError({
925
+ err: fallbackErr,
926
+ provider: fallbackResolved.parsed.provider,
927
+ model: fallbackResolved.parsed.model,
928
+ route: 'quota-model-fallback',
929
+ durationMs: Date.now() - fallbackStartMs,
930
+ });
931
+ const fallbackAttempts = quotaFallbackAttempts(runtimeAttemptsForError(normalizedFallbackErr, fallbackResolved.parsed.provider, fallbackResolved.parsed.model, 'quota-model-fallback'));
932
+ const attempts = resequenceRuntimeAttempts([...primaryAttempts, ...fallbackAttempts]);
933
+ const kind = normalizedFallbackErr.kind;
934
+ const fallbackMsg = normalizedFallbackErr.message;
935
+ throw new OrchestratorInvokeError(`Primary model '${rawModel}' failed and fallback model '${rawFallback}' also failed.\n\n` +
936
+ `Primary error:\n${primaryErr.message}\n\nFallback error:\n${fallbackMsg}`, kind, attempts, { cause: fallbackErr });
663
937
  }
664
938
  }
665
- else {
666
- throw err;
939
+ }
940
+ catch (err) {
941
+ if (opts.artifact !== undefined && err instanceof OrchestratorInvokeError) {
942
+ try {
943
+ const attempts = persistRuntimeAttempts(err.attempts, opts.customSecrets);
944
+ const shared = buildArtifactSharedFields({
945
+ artifact: opts.artifact,
946
+ safePrompt,
947
+ ...(safeSystemPrompt !== undefined ? { safeSystemPrompt } : {}),
948
+ ...(groundingBundle !== undefined ? { groundingBundle } : {}),
949
+ ...(opts.outputContract !== undefined ? { outputContract: opts.outputContract } : {}),
950
+ ...(opts.contextPolicy !== undefined ? { contextPolicy: opts.contextPolicy } : {}),
951
+ ...(opts.runMetadata !== undefined ? { runMetadata: opts.runMetadata } : {}),
952
+ });
953
+ const failureArtifact = {
954
+ schemaVersion: INVOCATION_FAILURE_ARTIFACT_SCHEMA_VERSION,
955
+ ...shared,
956
+ requestedBackend,
957
+ attempts,
958
+ terminal: {
959
+ kind: err.kind,
960
+ attempt: attempts.at(-1)?.sequence ?? 1,
961
+ message: persistRuntimeTextEvidence(runtimeMessageEvidence(err.message), opts.customSecrets, INVOKE_MESSAGE_EVIDENCE_LIMIT_BYTES),
962
+ },
963
+ createdAt: new Date().toISOString(),
964
+ };
965
+ const saved = saveInvocationFailureArtifact(path.join(configRoot, config.totemDir), failureArtifact);
966
+ err.failureArtifactHash = saved.hash;
967
+ log.dim(tag, `Invocation failure artifact ${saved.existed ? 'already recorded' : 'recorded'}: ${saved.hash.slice(0, 12)}…`);
968
+ opts.artifact.onFailureEmitted?.(saved.hash, saved.path);
969
+ // totem-context: companion failure evidence is warn-only by contract; preserve and rethrow the original invocation error after surfacing this emission failure.
970
+ }
971
+ catch (artifactErr) {
972
+ log.warn(tag, `Invocation failure artifact emission failed (original error preserved): ${artifactErr instanceof Error ? artifactErr.message : String(artifactErr)}`);
973
+ }
667
974
  }
975
+ throw err;
668
976
  }
669
977
  if (useCache && result.content && result.durationMs > 0) {
670
978
  try {
@@ -694,39 +1002,26 @@ export async function runOrchestrator(opts) {
694
1002
  // degradation is warned, never silent).
695
1003
  if (opts.artifact !== undefined) {
696
1004
  try {
697
- const inputBundle = {
698
- maskedPrompt: safePrompt,
699
- ...(safeSystemPrompt !== undefined && safeSystemPrompt.length > 0
700
- ? { maskedSystemPrompt: safeSystemPrompt }
701
- : {}),
702
- ...(opts.artifact.diffScope !== undefined ? { diffScope: opts.artifact.diffScope } : {}),
703
- ...(opts.artifact.specContract !== undefined
704
- ? { specContract: opts.artifact.specContract }
705
- : {}),
706
- };
1005
+ const shared = buildArtifactSharedFields({
1006
+ artifact: opts.artifact,
1007
+ safePrompt,
1008
+ ...(safeSystemPrompt !== undefined ? { safeSystemPrompt } : {}),
1009
+ ...(groundingBundle !== undefined ? { groundingBundle } : {}),
1010
+ ...(opts.outputContract !== undefined ? { outputContract: opts.outputContract } : {}),
1011
+ ...(opts.contextPolicy !== undefined ? { contextPolicy: opts.contextPolicy } : {}),
1012
+ ...(opts.runMetadata !== undefined ? { runMetadata: opts.runMetadata } : {}),
1013
+ });
707
1014
  const runArtifact = {
708
1015
  schemaVersion: RUN_ARTIFACT_SCHEMA_VERSION,
709
- inputBundle,
710
- inputHash: calculateDeterministicHash(inputBundle),
711
- grounding: {
712
- hash: opts.artifact.groundingHash,
713
- provenanceSummary: opts.artifact.provenanceSummary,
714
- // Verbatim passthrough — the bundle was assembled and hashed by the
715
- // caller (mmnto-ai/totem#2101); this seam records, never re-derives
716
- // or upgrades. Post-reconciliation (#2102): a caller-supplied
717
- // `groundingBundle` serves this role when `artifact.bundle` is absent.
718
- ...(groundingBundle !== undefined ? { bundle: groundingBundle } : {}),
719
- },
720
- backend: {
1016
+ ...shared,
1017
+ backend: buildArtifactBackend({
721
1018
  provider: resolved.parsed.provider,
722
1019
  model,
723
1020
  qualifiedModel,
724
- // #2102: caller-supplied wins; the default reproduces the slice-1
725
- // constant — every undeclared backend is factually completion-only.
726
1021
  admissionClass: requestedAdmissionClass,
727
1022
  taskProfile,
728
1023
  ...(opts.temperature !== undefined ? { temperature: opts.temperature } : {}),
729
- },
1024
+ }),
730
1025
  output: {
731
1026
  content: result.content,
732
1027
  metrics: {
@@ -738,23 +1033,14 @@ export async function runOrchestrator(opts) {
738
1033
  durationMs: result.durationMs,
739
1034
  ...(result.finishReason !== undefined ? { finishReason: result.finishReason } : {}),
740
1035
  },
1036
+ ...(result.attempts !== undefined
1037
+ ? {
1038
+ execution: {
1039
+ attempts: persistRuntimeAttempts(result.attempts, opts.customSecrets),
1040
+ },
1041
+ }
1042
+ : {}),
741
1043
  },
742
- // #2102: the admitted contract group is recorded ONLY when the caller
743
- // supplied at least one member — an omitted contract stays an omitted
744
- // key, never an empty object (additive 1.x, slice-1 artifacts unchanged).
745
- ...(opts.outputContract !== undefined ||
746
- opts.contextPolicy !== undefined ||
747
- opts.runMetadata !== undefined
748
- ? {
749
- admission: {
750
- ...(opts.outputContract !== undefined
751
- ? { outputContract: opts.outputContract }
752
- : {}),
753
- ...(opts.contextPolicy !== undefined ? { contextPolicy: opts.contextPolicy } : {}),
754
- ...(opts.runMetadata !== undefined ? { runMetadata: opts.runMetadata } : {}),
755
- },
756
- }
757
- : {}),
758
1044
  createdAt: new Date().toISOString(),
759
1045
  };
760
1046
  const saved = saveRunArtifact(path.join(configRoot, config.totemDir), runArtifact);