omp-conductor 0.15.12 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/REFERENCE.md +81 -6
- package/package.json +2 -1
- package/schema/config.schema.json +6 -0
- package/src/admission.ts +745 -0
- package/src/ask.ts +47 -0
- package/src/backups.ts +19 -7
- package/src/board.ts +1 -2
- package/src/briefs/orchestrator.md +62 -4
- package/src/cli.ts +26 -0
- package/src/commands/context.ts +3 -0
- package/src/commands/decision.ts +10 -1
- package/src/commands/doctor.ts +2 -0
- package/src/commands/message.ts +8 -1
- package/src/commands/restart.ts +93 -54
- package/src/commands/restore-db.ts +146 -0
- package/src/commands/stop.ts +66 -34
- package/src/commands/unfreeze.ts +56 -0
- package/src/commands/watch.ts +77 -0
- package/src/config-schema.ts +9 -0
- package/src/config.ts +24 -0
- package/src/daemon.ts +485 -577
- package/src/dashboard/server.ts +2 -1
- package/src/decisions.ts +32 -7
- package/src/depends-on.ts +73 -0
- package/src/doctor.ts +418 -8
- package/src/escalate.ts +122 -15
- package/src/failure-class.ts +47 -0
- package/src/fleet.ts +55 -377
- package/src/gitops.ts +86 -1
- package/src/lifecycle.ts +113 -2
- package/src/log.ts +40 -0
- package/src/model-fallback.ts +3 -2
- package/src/omp-settings.ts +114 -0
- package/src/omp.ts +63 -0
- package/src/orchestrator-down.ts +231 -0
- package/src/orchestrator-tick.ts +14 -1
- package/src/orchestrator.ts +14 -0
- package/src/release-policy.ts +163 -18
- package/src/reports.ts +124 -12
- package/src/session-host.ts +6 -0
- package/src/setup-host.ts +386 -17
- package/src/setup-install.ts +40 -2
- package/src/setup-wizard.ts +314 -113
- package/src/setup.ts +58 -1
- package/src/status-render.ts +445 -0
- package/src/stop-provenance.ts +119 -0
- package/src/store.ts +533 -11
- package/src/types.ts +298 -4
- package/src/unblock.ts +1 -1
- package/src/upgrade-verify.ts +1 -1
- package/src/upgrade.ts +27 -8
- package/src/verbs/protocol.ts +16 -3
- package/src/verbs/server.ts +52 -1
- package/src/wizard-ui.ts +261 -46
- package/src/worker.ts +183 -10
package/src/reports.ts
CHANGED
|
@@ -37,7 +37,14 @@
|
|
|
37
37
|
*/
|
|
38
38
|
|
|
39
39
|
import { availabilityDisposition, interruptDisposition } from "./availability.ts";
|
|
40
|
-
import {
|
|
40
|
+
import {
|
|
41
|
+
readTelegramToken,
|
|
42
|
+
resolveProjectTopicId,
|
|
43
|
+
sendTelegram,
|
|
44
|
+
telegramTextParts,
|
|
45
|
+
TelegramSendError,
|
|
46
|
+
} from "./escalate.ts";
|
|
47
|
+
import { createHash } from "node:crypto";
|
|
41
48
|
import { localDayKey } from "./digest-schedule.ts";
|
|
42
49
|
import {
|
|
43
50
|
DIGEST_BACKLOG_LIMIT,
|
|
@@ -86,11 +93,68 @@ const NO_ISSUE = 0;
|
|
|
86
93
|
const ERROR_SAMPLE = 90;
|
|
87
94
|
|
|
88
95
|
/**
|
|
89
|
-
*
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
*
|
|
96
|
+
* The within-run harness reliability a settlement now names where a human
|
|
97
|
+
* reads it (#584). The row always carries the resolved model/provider and the
|
|
98
|
+
* counts (written at settlement straight off the worker result); this is only
|
|
99
|
+
* the *wording* — the sentence a settlement report and a tick digest tell an
|
|
100
|
+
* operator who would otherwise have to read a transcript to learn that the
|
|
101
|
+
* run finished on a fallback after a mid-run swap, or rode out a throttled
|
|
102
|
+
* provider on retries and compaction.
|
|
103
|
+
*
|
|
104
|
+
* Deliberately empty for a clean run: a run that never swapped, retried or
|
|
105
|
+
* compacted changes neither the settlement report nor the digest, so the
|
|
106
|
+
* "additive" claim is the observable one — byte-for-byte today's output on a
|
|
107
|
+
* quiet fleet. The resolved model is still recorded on the row every time (#584's
|
|
108
|
+
* second fake), but a clean run's narrative has nothing to add.
|
|
109
|
+
*/
|
|
110
|
+
export interface ReliabilitySettlement {
|
|
111
|
+
resolvedModel?: string;
|
|
112
|
+
resolvedProvider?: string;
|
|
113
|
+
retryFallbacks: { from: string; to: string }[];
|
|
114
|
+
retryFallbackSucceeded: number;
|
|
115
|
+
modelRecoveries: number;
|
|
116
|
+
autoRetryCount: number;
|
|
117
|
+
autoCompactionCount: number;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* The human sentence for the above reliability surface, or undefined when there
|
|
122
|
+
* is nothing worth saying (a clean run). One line so it drops into both the
|
|
123
|
+
* settlement report and a material event's summary unchanged.
|
|
124
|
+
*/
|
|
125
|
+
export function reliabilitySettlementLine(
|
|
126
|
+
f: ReliabilitySettlement,
|
|
127
|
+
): string | undefined {
|
|
128
|
+
const { resolvedModel, resolvedProvider } = f;
|
|
129
|
+
if (f.retryFallbacks.length === 0 && f.retryFallbackSucceeded === 0 && f.modelRecoveries === 0 && f.autoRetryCount === 0 && f.autoCompactionCount === 0) {
|
|
130
|
+
return undefined;
|
|
131
|
+
}
|
|
132
|
+
const activity = [
|
|
133
|
+
f.retryFallbacks.length > 0
|
|
134
|
+
? `${f.retryFallbacks.length} within-run model swap${f.retryFallbacks.length === 1 ? "" : "s"}`
|
|
135
|
+
: undefined,
|
|
136
|
+
f.retryFallbackSucceeded > 0 ? `${f.retryFallbackSucceeded} recovered` : undefined,
|
|
137
|
+
f.modelRecoveries > 0 ? `${f.modelRecoveries} model recovery` : undefined,
|
|
138
|
+
f.autoRetryCount > 0 ? `${f.autoRetryCount} provider retry` : undefined,
|
|
139
|
+
f.autoCompactionCount > 0 ? `${f.autoCompactionCount} compaction` : undefined,
|
|
140
|
+
].filter((part): part is string => part !== undefined);
|
|
141
|
+
const finished =
|
|
142
|
+
resolvedModel === undefined
|
|
143
|
+
? "finished on the fallback"
|
|
144
|
+
: `finished on ${resolvedModel}${resolvedProvider === undefined ? "" : ` (${resolvedProvider})`}`;
|
|
145
|
+
return `Reliability: ${activity.join(", ")}; ${finished}.`;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* One delivery attempt's transport for a *single message*. Injected so the
|
|
150
|
+
* delivery contract — claim, split, send, record progress, retry — is testable
|
|
151
|
+
* without a live bot, which is the same split `confineToolCall` and `verifyPr`
|
|
152
|
+
* use: the decision is separable from the I/O. The outbox splits a long report
|
|
153
|
+
* into labelled parts itself and drives one call per part (so partial progress
|
|
154
|
+
* can be persisted and resumed, #566); a text passed here is always under the
|
|
155
|
+
* wire limit. Resolves with Telegram's own message id when it returns one,
|
|
156
|
+
* throws on any known failure — {@link TelegramSendError} carries the
|
|
157
|
+
* definitive/unknown verdict the retry's honesty hangs on.
|
|
94
158
|
*/
|
|
95
159
|
export type ReportSend = (text: string) => Promise<number | undefined>;
|
|
96
160
|
|
|
@@ -291,7 +355,14 @@ export function telegramReportSend(p: ProjectConfig): ReportSend {
|
|
|
291
355
|
"definitive",
|
|
292
356
|
);
|
|
293
357
|
}
|
|
294
|
-
|
|
358
|
+
// `sendTelegram` splits internally and resolves with every part's id. A
|
|
359
|
+
// caller on this seam always passes one message under the wire limit, so a
|
|
360
|
+
// single-element array comes back; `[0]` is that one id. A text over the
|
|
361
|
+
// limit would still be delivered whole (never truncated) — the seam would
|
|
362
|
+
// simply under-report the extra ids, which is why the split lives at the
|
|
363
|
+
// caller, not here.
|
|
364
|
+
const ids = await sendTelegram(token, chatId, text, { topicId: resolveProjectTopicId(p) });
|
|
365
|
+
return ids[0];
|
|
295
366
|
};
|
|
296
367
|
}
|
|
297
368
|
|
|
@@ -345,7 +416,13 @@ export async function deliverOperatorMessage(
|
|
|
345
416
|
});
|
|
346
417
|
return { kind: "held", category, noticeId: deps.noticeId, reason: disposition };
|
|
347
418
|
}
|
|
348
|
-
|
|
419
|
+
const send = deps.send ?? telegramReportSend(project);
|
|
420
|
+
// A long operator message is split and driven part-by-part through the same
|
|
421
|
+
// seam the outbox uses, so a custom transport sees the same per-part calls
|
|
422
|
+
// and the default one never truncates (#566).
|
|
423
|
+
for (const part of telegramTextParts(text)) {
|
|
424
|
+
await send(part);
|
|
425
|
+
}
|
|
349
426
|
return { kind: "sent", category };
|
|
350
427
|
}
|
|
351
428
|
|
|
@@ -596,9 +673,38 @@ export function createReportOutbox(deps: ReportOutboxDeps): ReportOutbox {
|
|
|
596
673
|
return;
|
|
597
674
|
}
|
|
598
675
|
|
|
599
|
-
|
|
676
|
+
const text = formatReportMessage(claimed, project.name);
|
|
677
|
+
const parts = telegramTextParts(text);
|
|
678
|
+
// Fingerprint of the exact text the split below is computed from. The
|
|
679
|
+
// per-part watermark on the row is only meaningful while this is unchanged:
|
|
680
|
+
// the resume point is a count into a specific split, so any drift in the
|
|
681
|
+
// text invalidates it (#566).
|
|
682
|
+
const messageHash = createHash("sha256").update(text).digest("hex");
|
|
683
|
+
// Resume at the first part this attempt has not yet confirmed. A count is
|
|
684
|
+
// only a valid resume point while the text — and therefore the split — is
|
|
685
|
+
// byte-identical to the attempt that wrote the watermark. Any change,
|
|
686
|
+
// notably the possible-repeat banner an unknown outcome adds to an
|
|
687
|
+
// ambiguous report, reshapes the parts, so the watermark is deliberately
|
|
688
|
+
// not carried across it and the whole report is re-sent amber-flagged —
|
|
689
|
+
// the same at-least-once semantics as the single-message path (#566).
|
|
690
|
+
const resumeAt = claimed.sentPartsHash === messageHash ? (claimed.sentParts ?? 0) : 0;
|
|
691
|
+
const start = Math.min(resumeAt, parts.length);
|
|
692
|
+
|
|
693
|
+
const messageIds: number[] = [];
|
|
600
694
|
try {
|
|
601
|
-
|
|
695
|
+
// Strictly in order, one `sendMessage` at a time: parts never interleave
|
|
696
|
+
// with each other or with a message sent concurrently to the same chat.
|
|
697
|
+
for (let i = start; i < parts.length; i += 1) {
|
|
698
|
+
const id = await send(parts[i]!);
|
|
699
|
+
if (id !== undefined) messageIds.push(id);
|
|
700
|
+
// Persist progress before the next part ships. Only a multi-part send
|
|
701
|
+
// has a crossing point to protect: a crash or a definitive failure can
|
|
702
|
+
// then only ever re-send the part that was in flight, never the
|
|
703
|
+
// confirmed prefix. Guarded by attempt id like every other transition.
|
|
704
|
+
if (parts.length > 1) {
|
|
705
|
+
store.markReportPartsSent(claimed.id, attemptId, i + 1, messageHash, now());
|
|
706
|
+
}
|
|
707
|
+
}
|
|
602
708
|
} catch (err) {
|
|
603
709
|
const at = now();
|
|
604
710
|
const error = err instanceof Error ? err.message : String(err);
|
|
@@ -641,15 +747,21 @@ export function createReportOutbox(deps: ReportOutboxDeps): ReportOutbox {
|
|
|
641
747
|
pass.requeued.push(claimed.id);
|
|
642
748
|
log(
|
|
643
749
|
`report ${claimed.id} attempt ${claimed.attempts}/${REPORT_MAX_ATTEMPTS} was rejected, ` +
|
|
750
|
+
`${start + messageIds.length}/${parts.length} parts accepted, ` +
|
|
644
751
|
`retrying in ${Math.round(delay / 1_000)}s: ${error}`,
|
|
645
752
|
);
|
|
646
753
|
return;
|
|
647
754
|
}
|
|
648
755
|
|
|
649
|
-
store.markReportDelivered(claimed.id, attemptId,
|
|
756
|
+
store.markReportDelivered(claimed.id, attemptId, messageIds, now());
|
|
650
757
|
pass.delivered.push(claimed.id);
|
|
651
758
|
log(
|
|
652
|
-
`report ${claimed.id} delivered
|
|
759
|
+
`report ${claimed.id} delivered` +
|
|
760
|
+
(parts.length === 1
|
|
761
|
+
? messageIds[0] === undefined
|
|
762
|
+
? ""
|
|
763
|
+
: ` as telegram message ${messageIds[0]}`
|
|
764
|
+
: ` as ${parts.length} telegram messages (${messageIds.join(", ")})`) +
|
|
653
765
|
`${claimed.ambiguous ? " (marked as a possible repeat)" : ""}`,
|
|
654
766
|
);
|
|
655
767
|
};
|
package/src/session-host.ts
CHANGED
|
@@ -50,6 +50,11 @@ export interface SessionHostSpec {
|
|
|
50
50
|
verbSocketPath?: string;
|
|
51
51
|
/** Deny every tool but reading and searching (#307) — the setup probes. */
|
|
52
52
|
readOnly?: boolean;
|
|
53
|
+
/** The fleet-owned omp settings overlay (#537): absolute path to the YAML
|
|
54
|
+
* overlay the daemon materialised from the project's `ompSettings` map.
|
|
55
|
+
* Carried as a plain scalar like the other fields; absent, nothing is staged
|
|
56
|
+
* on the far side either. */
|
|
57
|
+
ompSettingsFile?: string;
|
|
53
58
|
}
|
|
54
59
|
|
|
55
60
|
/** Parent → child. */
|
|
@@ -186,6 +191,7 @@ export async function runSessionHost(
|
|
|
186
191
|
...(spec.releaseGrants === undefined ? {} : { releaseGrants: spec.releaseGrants }),
|
|
187
192
|
...(spec.verbSocketPath === undefined ? {} : { verbSocketPath: spec.verbSocketPath }),
|
|
188
193
|
...(spec.readOnly === undefined ? {} : { readOnly: spec.readOnly }),
|
|
194
|
+
...(spec.ompSettingsFile === undefined ? {} : { ompSettingsFile: spec.ompSettingsFile }),
|
|
189
195
|
// The release audit lives in the daemon's state directory, which this
|
|
190
196
|
// process may not be able to write and must not be trusted to. It
|
|
191
197
|
// becomes a message; the parent performs the durable write.
|