@staix/agent-hub 0.12.3 → 0.12.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/hub/report.ts CHANGED
@@ -102,6 +102,9 @@ export function summarize(events: StampedEvent[]): Report {
102
102
  case "undeliverable":
103
103
  r.messages[e.type]++;
104
104
  break;
105
+ case "stale": // published, then dropped before delivery (issue #106)
106
+ r.messages.dropped.stale = (r.messages.dropped.stale ?? 0) + 1;
107
+ break;
105
108
  case "turn_end":
106
109
  peer(e.peer).turns++;
107
110
  peer(e.peer).busyMinutes += e.ms / 60_000;
@@ -248,3 +248,88 @@ export function assign(
248
248
  if (owner === PI) trace.push(`pi decision: backend ${piBackend ?? "dgx"}${task.signals.includes("long_context") ? `, context limit ${piBackend === "mlx" ? routing.pi.mlx_max_context_tokens : routing.pi.dgx_max_context_tokens}` : ""}`);
249
249
  return { ...(owner ? { owner } : {}), ...(pii ? { reviewer: "user" } : reviewer ? { reviewer } : {}), ...(route ? { route } : {}), fixedModel, ...(piBackend ? { piBackend } : {}), trace };
250
250
  }
251
+
252
+ /**
253
+ * One recorded task of a peer in a class, as stage proxies (issue #109): `orient` from being handed the task to its
254
+ * accept (queueing included), `work` from the accept to its done. Unknown stages stay undefined, never zero.
255
+ */
256
+ export interface SplitObservation {
257
+ outcome: "approved" | "failed";
258
+ orient?: number;
259
+ work?: number;
260
+ }
261
+
262
+ /** Everything the shadow split prediction reads; `Tasks` builds it once and `route explain` shows the same. */
263
+ export interface SplitInput {
264
+ /** The pair: the peer routing would pick for the task, and the owner of the open task it overlaps. */
265
+ peers: [PeerId, PeerId];
266
+ /** Each peer's observations in this class, from this hub run only: one version, one hook profile. */
267
+ observations: Record<PeerId, SplitObservation[]>;
268
+ /** Work units of the routed task and of the one it overlaps; undefined is unknown. */
269
+ units: [number | undefined, number | undefined];
270
+ /** Other open work each peer has: a busy owner does not start from orientation plus two whole units. */
271
+ backlog: Record<PeerId, number>;
272
+ available: Record<PeerId, boolean>;
273
+ }
274
+
275
+ export interface SplitPrediction {
276
+ verdict: "split" | "single" | "unknown";
277
+ trace: string[];
278
+ /** The peer that would finish both units soonest on its own. */
279
+ single?: PeerId;
280
+ /** Predicted seconds: both peers one unit each, in parallel (the later of the two), and the best peer alone. */
281
+ splitS?: number;
282
+ singleS?: number;
283
+ }
284
+
285
+ /** Observations a peer needs, and the share of failures a history may hold, before a prediction is made. */
286
+ export const SPLIT_MIN = 5;
287
+ const SPLIT_FAILURE_SHARE = 0.3;
288
+
289
+ /**
290
+ * Shadow split prediction (issue #109): never changes assignment. For two similar units of overlapping work, a split
291
+ * (each peer one unit, in parallel) finishes when the later of `o + u` does; the best single peer takes `o + 2u`. With
292
+ * a faster and a slower peer that is `o_s + u_s < o_f + 2u_f`. It is a model assumption (equal units, measured
293
+ * orientation, free coordination), not a bound, so anything that breaks it makes the prediction unknown: unequal or
294
+ * unknown units, a busy or unavailable peer, too few or failure-heavy records, or work times too spread out to call
295
+ * comparable. A difference under a tenth of the single time is inconclusive.
296
+ */
297
+ export function predictSplit(input: SplitInput): SplitPrediction {
298
+ const trace: string[] = ["shadow split prediction (it never changes assignment):"];
299
+ const unknown = (why: string): SplitPrediction => ({ verdict: "unknown", trace: [...trace, ` unknown: ${why}`] });
300
+ const [a, b] = input.peers;
301
+ if (a === b) return unknown("the routed task and the one it overlaps have the same owner");
302
+ const [ua, ub] = input.units;
303
+ if (ua === undefined || ub === undefined || ua !== ub) return unknown(`work units ${ua ?? "?"} and ${ub ?? "?"}: the rule needs two equal, known units`);
304
+ for (const p of [a, b]) {
305
+ if (!input.available[p]) return unknown(`${p} is not available`);
306
+ if ((input.backlog[p] ?? 0) > 0) return unknown(`${p} has ${input.backlog[p]} other open task(s): it would not start from orientation plus two units`);
307
+ }
308
+ const median = (xs: number[]) => {
309
+ const v = [...xs].sort((x, y) => x - y);
310
+ return v.length % 2 ? v[(v.length - 1) / 2]! : (v[v.length / 2 - 1]! + v[v.length / 2]!) / 2;
311
+ };
312
+ const stats: Record<PeerId, { o: number; u: number }> = {};
313
+ for (const p of [a, b]) {
314
+ const all = input.observations[p] ?? [];
315
+ const failed = all.filter((o) => o.outcome !== "approved").length;
316
+ const ok = all.filter((o): o is Required<SplitObservation> => o.outcome === "approved" && o.orient !== undefined && o.work !== undefined);
317
+ if (ok.length < SPLIT_MIN) return unknown(`${p} has ${ok.length} measured task(s) in this hub run; ${SPLIT_MIN} are needed`);
318
+ if (failed / all.length > SPLIT_FAILURE_SHARE) return unknown(`${failed} of ${all.length} of ${p}'s recorded tasks failed: its successes alone would understate its time`);
319
+ const works = ok.map((o) => o.work).sort((x, y) => x - y);
320
+ const u = median(works);
321
+ const iqr = works[Math.floor((works.length * 3) / 4)]! - works[Math.floor(works.length / 4)]!;
322
+ if (u <= 0 || iqr / u > 1) return unknown(`${p}'s work times spread too widely (IQR ${Math.round(iqr / 1000)} s against a median of ${Math.round(u / 1000)} s) to call its tasks comparable units`);
323
+ stats[p] = { o: median(ok.map((o) => o.orient)), u };
324
+ trace.push(` ${p}: orientation ${Math.round(stats[p]!.o / 1000)} s, one unit ${Math.round(u / 1000)} s (median of ${ok.length})`);
325
+ }
326
+ const single = stats[a]!.o + 2 * stats[a]!.u <= stats[b]!.o + 2 * stats[b]!.u ? a : b;
327
+ const s = (ms: number) => Math.round(ms / 1000);
328
+ const splitS = s(Math.max(stats[a]!.o + stats[a]!.u, stats[b]!.o + stats[b]!.u));
329
+ const singleS = s(stats[single]!.o + 2 * stats[single]!.u);
330
+ trace.push(` split: ${splitS} s (${a} ${s(stats[a]!.o + stats[a]!.u)} s, ${b} ${s(stats[b]!.o + stats[b]!.u)} s for one unit each); ${single} alone: ${singleS} s for two`);
331
+ if (Math.abs(splitS - singleS) < singleS / 10) return { ...unknown("the difference is under a tenth of the single time: inconclusive"), single, splitS, singleS };
332
+ const verdict = splitS < singleS ? "split" : "single";
333
+ trace.push(verdict === "split" ? " a split would finish sooner" : ` ${single} alone would finish sooner`);
334
+ return { verdict, trace, single, splitS, singleS };
335
+ }