@nanobpm/nano-workforce 0.188.1 → 0.189.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/SPEC.md +16 -0
- package/app/adjudications.test.ts +735 -0
- package/app/adjudications.ts +378 -0
- package/app/agentCompletion.test.ts +282 -10
- package/app/agentCompletion.ts +163 -23
- package/app/agentic/cockpit/mount.test.ts +50 -0
- package/app/agentic/cockpit/supply-render.test.ts +20 -0
- package/app/agentic/cockpit/supply-render.ts +15 -0
- package/app/agentic/cockpit/supply-view.ts +19 -2
- package/app/agentic/permission-bridge.test.ts +2 -2
- package/app/agentic/vocab/demand-report.test.ts +66 -1
- package/app/agentic/vocab/demand-report.ts +54 -6
- package/app/answer-escalation.test.ts +415 -2
- package/app/answerContextMapping.test.ts +83 -0
- package/app/contracts.ts +24 -0
- package/app/convergenceAdjudicationResume.test.ts +274 -0
- package/app/github.ts +10 -0
- package/app/harnessProtocol.test.ts +170 -0
- package/app/harnessProtocol.ts +312 -0
- package/app/mcpToolSurface.ts +7 -1
- package/app/service.test.ts +178 -1
- package/app/service.ts +104 -4
- package/app/terminalReaderBehaviour.test.ts +21 -0
- package/db/migrations/107_worker_harness_protocol.sql +30 -0
- package/db/migrations/109_pr_adjudications.sql +61 -0
- package/db/migrations/110_task_completions_auto_applied.sql +34 -0
- package/openapi.yaml +71 -1
- package/operations/completeUserTask.test.ts +5 -5
- package/operations/enrolAgenticWorker.test.ts +84 -0
- package/operations/enrolAgenticWorker.ts +67 -7
- package/operations/getAgenticRegistry.ts +1 -1
- package/operations/getAgenticSupply.test.ts +80 -0
- package/operations/getAgenticSupply.ts +15 -3
- package/operations/listEscalations.test.ts +1 -1
- package/package.json +1 -1
- package/pages/cockpit/mount.js +18 -0
- package/resources/processes/convergence-loop.bpmn +9 -0
- package/resources/processes/merge-loop.bpmn +1 -0
- package/test/worldDb.ts +6 -0
- package/workers/answer-escalation/worker.ts +191 -11
package/app/service.ts
CHANGED
|
@@ -9,7 +9,8 @@
|
|
|
9
9
|
// `Table<T>` surface), not hand-written SQL. Row shapes are declared inline here.
|
|
10
10
|
import type { DataLayer, EngineClient } from "@nanobpm/urban";
|
|
11
11
|
import { ABANDONED_STATUS, abandonUrl, mintAbandonToken, renderAbandonBrief } from "./abandon.ts";
|
|
12
|
-
import {
|
|
12
|
+
import { matchAdjudication, prAdjudications, resetAdjudications } from "./adjudications.ts";
|
|
13
|
+
import { completeEscalationAutoApplied, escalationFormId } from "./agentCompletion.ts";
|
|
13
14
|
import { agentSlaTimeout } from "./agentSla.ts";
|
|
14
15
|
import {
|
|
15
16
|
CAPS_RESOLVED_MESSAGE,
|
|
@@ -585,6 +586,14 @@ export async function submitPr(
|
|
|
585
586
|
for (const e of await escs(data).find({ pr_key: parsed.prKey, status: "open" })) {
|
|
586
587
|
await escs(data).update(e.id, { status: "stale" });
|
|
587
588
|
}
|
|
589
|
+
// A fresh convergence run must ALSO start with a clean durable adjudication memory (issue #806,
|
|
590
|
+
// Copilot review): the auto-resume replays a prior `(PR, question)` answer forever, so a re-opened
|
|
591
|
+
// PR whose question recurs would silently auto-apply the stale decision and an operator could never
|
|
592
|
+
// force a fresh one. This PR's adjudications are invalidated on reopen — but the reset is deferred
|
|
593
|
+
// to AFTER `process_key` is advanced to the new instance (see below), NOT here: clearing the memory
|
|
594
|
+
// while `process_key` still names the OLD instance leaves a window where a delayed old-instance
|
|
595
|
+
// answer job still passes the worker's staleness gate and reinserts its adjudication into the fresh
|
|
596
|
+
// run (Copilot review of #806). Advancing the run identity FIRST, then clearing, fences that job.
|
|
588
597
|
// Re-open a previously converged/abandoned/merged PR for a fresh convergence run.
|
|
589
598
|
await table.update(parsed.prKey, {
|
|
590
599
|
status: "converging",
|
|
@@ -682,8 +691,44 @@ export async function submitPr(
|
|
|
682
691
|
},
|
|
683
692
|
});
|
|
684
693
|
const processKey = processInstanceKey == null ? null : String(processInstanceKey);
|
|
685
|
-
|
|
686
|
-
|
|
694
|
+
// The `process_key` advance and the adjudication reset below are the two writes that MAKE the new
|
|
695
|
+
// run authoritative. If EITHER throws, the newly created instance is already live but the reopen is
|
|
696
|
+
// only half-committed — and a retry would short-circuit at the `alreadyRunning` idempotency gate
|
|
697
|
+
// (the new instance is ACTIVE, so `derived_status` is non-terminal), never re-running the reset. A
|
|
698
|
+
// failed reset would then leave the fresh run replaying STALE adjudication memory indefinitely
|
|
699
|
+
// (Copilot review). So roll the just-created run back on failure: terminate it and rethrow, so the
|
|
700
|
+
// submission is NOT treated as started. Terminating flips the PR's derived tracking status to a
|
|
701
|
+
// terminal edge (`abandoned`) via the `instanceTracking` reconciler, making the PR resubmittable so
|
|
702
|
+
// a retry re-creates a fresh instance and re-runs the reset cleanly — no orphaned run auto-applies
|
|
703
|
+
// stale decisions in the meantime.
|
|
704
|
+
try {
|
|
705
|
+
if (processKey != null) {
|
|
706
|
+
await table.update(parsed.prKey, { process_key: processKey });
|
|
707
|
+
}
|
|
708
|
+
// Invalidate this PR's durable adjudication memory for the fresh run (issue #806, Copilot review) —
|
|
709
|
+
// deferred to HERE, after `process_key` is advanced to the new instance above, so the reset happens
|
|
710
|
+
// UNDER the new run identity. On reopen (`existing`), any delayed old-instance answer job is now
|
|
711
|
+
// rejected by the worker's staleness gate (its `processInstanceKey` no longer matches the advanced
|
|
712
|
+
// `process_key`), so it cannot reinsert a stale adjudication after the reset; and the worker reads
|
|
713
|
+
// `process_key` as late as possible so it observes this advance. The insert-if-absent record then
|
|
714
|
+
// re-learns the operator's new answer for the new run. Runs unconditionally (even if `processKey` is
|
|
715
|
+
// null: the memory must still be clean for the fresh run). The wipe is a SINGLE atomic `DELETE`
|
|
716
|
+
// (`resetAdjudications`), never a row-by-row loop, so a crash mid-reset cannot leave a partially
|
|
717
|
+
// cleared memory (Copilot review of #806).
|
|
718
|
+
if (existing) {
|
|
719
|
+
await resetAdjudications(data, parsed.prKey);
|
|
720
|
+
}
|
|
721
|
+
} catch (err) {
|
|
722
|
+
if (processInstanceKey != null) {
|
|
723
|
+
try {
|
|
724
|
+
await engine.cancelInstance({ processInstanceKey: String(processInstanceKey) });
|
|
725
|
+
} catch (cancelErr) {
|
|
726
|
+
// Best-effort: a failed rollback-cancel leaves the instance for the abandon/reconcile poller
|
|
727
|
+
// to reap, but must not mask the original error that the caller needs to see and retry on.
|
|
728
|
+
console.warn(`[submit] ${parsed.prKey} rollback-cancel of ${processInstanceKey} failed: ${cancelErr}`);
|
|
729
|
+
}
|
|
730
|
+
}
|
|
731
|
+
throw err;
|
|
687
732
|
}
|
|
688
733
|
return { prKey: parsed.prKey, processKey };
|
|
689
734
|
}
|
|
@@ -2839,12 +2884,67 @@ export async function pollUserTasks(
|
|
|
2839
2884
|
// Desired set, deduped by completable key (a task is open at most once; guard a page overlap / a
|
|
2840
2885
|
// subject seen under two statuses mid-pass).
|
|
2841
2886
|
const desiredByKey = new Map<string, UserTaskRow>();
|
|
2887
|
+
// Keys auto-resumed from a durable adjudication this pass (issue #806). The reduced-capability scan
|
|
2888
|
+
// visits each instance twice (direct + callActivity hierarchy), and both queries snapshot the task
|
|
2889
|
+
// BEFORE the resume removes it, so the second visit would otherwise re-attempt a now-gone completion
|
|
2890
|
+
// and fall through to projecting the very row we just retired. Recording the key keeps the resume
|
|
2891
|
+
// one-shot and out of the inbox.
|
|
2892
|
+
const resumedByKey = new Set<string>();
|
|
2842
2893
|
const project = async (elementId: string | undefined, userTaskKey: string, processInstanceKey: string, rootProcessInstanceKey: string, formKey: string) => {
|
|
2843
2894
|
if (!elementId) return;
|
|
2844
2895
|
const rowKey = userTaskKey.trim();
|
|
2845
|
-
if (!rowKey || desiredByKey.has(rowKey)) return;
|
|
2896
|
+
if (!rowKey || desiredByKey.has(rowKey) || resumedByKey.has(rowKey)) return;
|
|
2846
2897
|
const ctx = await contextFor(elementId, userTaskKey, processInstanceKey, rootProcessInstanceKey, formKey);
|
|
2847
2898
|
if (!ctx) return;
|
|
2899
|
+
// Durable adjudication auto-resume (issue #806): before surfacing a NEW convergence `wait-answer`
|
|
2900
|
+
// to a human, check whether THIS PR already has a settled adjudication for the SAME question
|
|
2901
|
+
// (canonical `questionFingerprint`). If it does, resume the loop with the recorded answer through
|
|
2902
|
+
// the canonical `completeUserTaskAttributed` door — attributed to the prior adjudicator and marked
|
|
2903
|
+
// `auto_applied` (a machine replay, reversible so a human can still override) so it is never
|
|
2904
|
+
// laundered into a first-hand irreversible human authority (Copilot review of #806) — instead of
|
|
2905
|
+
// re-parking a human on an already-answered question (PR #800 / proc 46310: the same design
|
|
2906
|
+
// question escalated at round 2 and again at round 13). Scoped to the review loop's `wait-answer`
|
|
2907
|
+
// on a real PR key; on any resolution failure the task still projects, so an un-resumable question
|
|
2908
|
+
// always reaches a human (fail-open to the human).
|
|
2909
|
+
if (elementId === PR_WAIT_ANSWER_ELEMENT && ctx.subjectType === "pr" && ctx.question && parsePr(ctx.subjectKey)) {
|
|
2910
|
+
try {
|
|
2911
|
+
// The adjudication LOOKUP lives inside this fail-open `try` (not just the resume) so a transient
|
|
2912
|
+
// `pr_adjudications.find` error never rejects `project` and aborts `pollUserTasks` mid-pass — the
|
|
2913
|
+
// task still projects and the question always reaches a human (SPEC: adjudication-resolution
|
|
2914
|
+
// failures fail open to the human).
|
|
2915
|
+
const adjudication = matchAdjudication(await prAdjudications(data).find({ pr_key: ctx.subjectKey }), ctx.question);
|
|
2916
|
+
// Only auto-resume when the prior adjudicator's provenance is KNOWN. A settled row with a blank
|
|
2917
|
+
// `adjudicated_by` (completed out of band, so `latestAdjudicator` returned no actor) must NOT be
|
|
2918
|
+
// manufactured into a synthetic `human` actor — that would audit an unknown-provenance replay as
|
|
2919
|
+
// a first-hand human decision. Fail open to a fresh human task instead (Copilot review of #806).
|
|
2920
|
+
const adjudicatedBy = adjudication?.adjudicated_by?.trim();
|
|
2921
|
+
if (adjudication && adjudicatedBy) {
|
|
2922
|
+
const resumed = await completeEscalationAutoApplied(data, engine, {
|
|
2923
|
+
userTaskKey: rowKey,
|
|
2924
|
+
// The sweep already discovered this task's owning instance — hand it to the resolve so the
|
|
2925
|
+
// auto-apply scans that ONE instance, not every open user task engine-wide (issue #806
|
|
2926
|
+
// Copilot review: an unfiltered per-task scan makes a single poll pass O(N²) across N
|
|
2927
|
+
// already-adjudicated PRs). `contextFor`/the sweep report the task's direct instance, so the
|
|
2928
|
+
// filtered scan finds exactly this task; a miss still fails open to the human.
|
|
2929
|
+
processInstanceKey,
|
|
2930
|
+
variables: { answer: adjudication.answer },
|
|
2931
|
+
actor: {
|
|
2932
|
+
kind: adjudication.adjudicated_kind === "agent" ? "agent" : "human",
|
|
2933
|
+
id: adjudicatedBy,
|
|
2934
|
+
},
|
|
2935
|
+
// Link the auto-apply back to the replayed adjudication (issue #806) so a human revert of the
|
|
2936
|
+
// resulting completion invalidates this exact decision instead of it being silently re-applied.
|
|
2937
|
+
adjudicationId: adjudication.id,
|
|
2938
|
+
});
|
|
2939
|
+
if (resumed.ok) {
|
|
2940
|
+
resumedByKey.add(rowKey);
|
|
2941
|
+
return;
|
|
2942
|
+
}
|
|
2943
|
+
}
|
|
2944
|
+
} catch (err) {
|
|
2945
|
+
console.error(`[poller] adjudication auto-resume (${ctx.subjectKey}): ${err}`);
|
|
2946
|
+
}
|
|
2947
|
+
}
|
|
2848
2948
|
const row = buildUserTaskRow(ctx, at);
|
|
2849
2949
|
if (row) desiredByKey.set(rowKey, row);
|
|
2850
2950
|
};
|
|
@@ -51,6 +51,27 @@ function memData(stores: Stores) {
|
|
|
51
51
|
return {
|
|
52
52
|
table: withTrackingViews((name: string, key: string) =>
|
|
53
53
|
memTable(stores[name]?.rows ?? [], stores[name]?.key ?? key)),
|
|
54
|
+
// Emulates the atomic bulk `DELETE FROM "pr_adjudications" WHERE "pr_key" = ?` submitPr issues via
|
|
55
|
+
// `data.open().exec` to reset a reopened PR's adjudication memory (Copilot review of #806). The SQL
|
|
56
|
+
// is validated against real SQLite in app/adjudications.test.ts; here it need only mutate the store.
|
|
57
|
+
open: () => ({
|
|
58
|
+
exec: async (sql: string, params: any[] = []) => {
|
|
59
|
+
if (/DELETE FROM "pr_adjudications" WHERE "pr_key" = \?/.test(sql)) {
|
|
60
|
+
const store = stores.pr_adjudications;
|
|
61
|
+
let changed = 0;
|
|
62
|
+
if (store) {
|
|
63
|
+
for (let i = store.rows.length - 1; i >= 0; i--) {
|
|
64
|
+
if (store.rows[i].pr_key === params[0]) {
|
|
65
|
+
store.rows.splice(i, 1);
|
|
66
|
+
changed++;
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
return { changed };
|
|
71
|
+
}
|
|
72
|
+
throw new Error(`unexpected exec sql: ${sql}`);
|
|
73
|
+
},
|
|
74
|
+
}),
|
|
54
75
|
} as any;
|
|
55
76
|
}
|
|
56
77
|
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
-- 107_worker_harness_protocol.sql — issue #802: make stale worker harnesses observable and gateable.
|
|
2
|
+
--
|
|
3
|
+
-- A stale worker harness (a `c8ctl-nano` build predating the AgentInstance-minting + transcript-flush
|
|
4
|
+
-- + result-envelope path) silently services jobs and swallows every machine-readable artifact, so good
|
|
5
|
+
-- agent work is lost to the orchestration and every run it touches dead-ends at a human (#796/#801).
|
|
6
|
+
-- Job routing is blind to harness capability/version, so a stale harness wins job leases
|
|
7
|
+
-- indistinguishably from a healthy one.
|
|
8
|
+
--
|
|
9
|
+
-- The fix persists the protocol version a harness advertises at ENROLMENT — a WORKER ATTRIBUTE (ADR
|
|
10
|
+
-- 0056 §7 — capability gates enrolment, NEVER a routing token `network.role#seat`) — so the app can
|
|
11
|
+
-- (a) expose it in `getAgenticSupply` / the registry and (b) flag RED / refuse agent-job routing for
|
|
12
|
+
-- any worker below the minimum protocol. A MISSING version is treated as stale.
|
|
13
|
+
--
|
|
14
|
+
-- This mirrors `worker_durable_resume` (migration 052): one FK-free table keyed by the worker
|
|
15
|
+
-- instance (`register.instance` / the enrol `instance`). FK-free by design — enrolment is per-worker
|
|
16
|
+
-- and connection-agnostic, with no parent row to reference. EXPAND (additive) phase: one new table;
|
|
17
|
+
-- nothing is dropped or renamed. Migration-prefix block 107–108 pre-assigned to this slice
|
|
18
|
+
-- (issue #802) off the origin/main high-water mark (104). The runner wraps each file in its own
|
|
19
|
+
-- transaction, so this file must NOT contain BEGIN/COMMIT.
|
|
20
|
+
|
|
21
|
+
CREATE TABLE IF NOT EXISTS worker_harness_protocol (
|
|
22
|
+
instance TEXT PRIMARY KEY, -- the worker instance id (enrol `instance` / register.instance)
|
|
23
|
+
-- The protocol version the harness advertised, or NULL when it advertised NONE. NULL is a first-class
|
|
24
|
+
-- "advertised no version" marker — distinct in intent from an absent row (never enrolled here), but
|
|
25
|
+
-- BOTH resolve to STALE at read time (absent version = stale). Recorded even on a downgrade so a
|
|
26
|
+
-- harness that previously advertised a healthy protocol and later re-enrols WITHOUT one clears its
|
|
27
|
+
-- stale-healthy value (mirrors the worker_durable_resume degrade-to-scratch semantics).
|
|
28
|
+
harness_protocol INTEGER,
|
|
29
|
+
updated_at TEXT NOT NULL
|
|
30
|
+
);
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
-- 109_pr_adjudications.sql — issue #806: persist wait-answer human adjudications so an
|
|
2
|
+
-- already-answered convergence question does not re-escalate.
|
|
3
|
+
--
|
|
4
|
+
-- The convergence loop's `wait-answer` escalation (`record-answer` → resume) applies a human's
|
|
5
|
+
-- answer to the CURRENT round but keeps NO durable memory that "(this PR, this question) was already
|
|
6
|
+
-- adjudicated to X". A stateless later round that re-derives the identical escalation condition
|
|
7
|
+
-- re-parks a human from scratch (PR #800 / proc 46310: the same design question escalated at round 2
|
|
8
|
+
-- and again at round 13, both answered identically). A human adjudication is NOT GitHub-derivable —
|
|
9
|
+
-- it is a fact the app itself must remember — so it lives here, durably.
|
|
10
|
+
--
|
|
11
|
+
-- One row per (PR, question) a human has settled: `pr.answer-escalation` (record-answer) writes it on
|
|
12
|
+
-- answering, and the poller (`pollUserTasks`) reads it before surfacing a NEW `wait-answer` — when the
|
|
13
|
+
-- question's fingerprint matches an existing row it auto-resumes with the recorded answer (attributed
|
|
14
|
+
-- to the prior adjudicator) instead of re-escalating a human.
|
|
15
|
+
--
|
|
16
|
+
-- • question_fingerprint — the canonical `normalizeAdvisoryText` + `fingerprint` digest of the
|
|
17
|
+
-- escalation question (app/github.ts `questionFingerprint`), the SAME line-stable normalisation
|
|
18
|
+
-- advisory acks use; so only a byte/semantic-identical, already-answered question is suppressed
|
|
19
|
+
-- while a materially different question still escalates. No second fingerprint implementation.
|
|
20
|
+
-- • answer / adjudicated_by / adjudicated_kind / adjudicated_at — the settled answer, who settled it,
|
|
21
|
+
-- whether they were a `human` or an `agent` (ADR 0046), and when, so the auto-resume replays the
|
|
22
|
+
-- exact decision AND preserves the original attribution kind — a human-settled decision replays as
|
|
23
|
+
-- human, an agent-settled one as agent, so an auto-apply can never launder an agent decision into an
|
|
24
|
+
-- irreversible human authority (Copilot review of #806).
|
|
25
|
+
-- • invalidated_at — a TOMBSTONE set when a human REVERTS the auto-applied completion that replayed
|
|
26
|
+
-- this decision (`revertAgentCompletion` → `invalidateAdjudication`, Copilot review of #806). A plain
|
|
27
|
+
-- DELETE is NOT race-safe: the reverted completion's `record-answer` job can be redelivered
|
|
28
|
+
-- (at-least-once) AFTER the delete and re-insert the SAME `(pr_key, question_fingerprint)`, so the
|
|
29
|
+
-- next poller pass re-auto-applies and silently undoes the revert. Keeping the row as a tombstone lets
|
|
30
|
+
-- the `UNIQUE (pr_key, question_fingerprint)` fence make that redelivered re-insert a no-op, and
|
|
31
|
+
-- `matchAdjudication` skips a tombstoned row so it never auto-applies again. The tombstone is cleared
|
|
32
|
+
-- only by `resetAdjudications` on a fresh-run re-submit. NULL for a live, replayable decision.
|
|
33
|
+
-- • source_completion_id — the `task_completions.id` of the WINNING completion that produced this
|
|
34
|
+
-- decision (issue #806 review). A FIRST-HAND agent answer to a `wait-answer` records its own durable
|
|
35
|
+
-- adjudication (`adjudicated_kind="agent"`) but — unlike a machine auto-apply — its ledger row has
|
|
36
|
+
-- `auto_applied=0` and NO `source_adjudication_id`, so a human revert of that reversible agent
|
|
37
|
+
-- completion could not previously find and tombstone the decision it created, and the poller would
|
|
38
|
+
-- re-auto-apply the reverted answer. Linking every convergence adjudication to its winning completion
|
|
39
|
+
-- lets `revertAgentCompletion` invalidate the decision on ANY reversible agent revert, not only a
|
|
40
|
+
-- machine replay (`invalidateAdjudicationByCompletion`). INSERT-if-absent, so only the ORIGINAL
|
|
41
|
+
-- first-hand completion is recorded; a later auto-apply's re-record is a UNIQUE no-op that leaves the
|
|
42
|
+
-- link pointing at the first-hand winner. NULL for a legacy/uncorrelated answer.
|
|
43
|
+
--
|
|
44
|
+
-- `UNIQUE (pr_key, question_fingerprint)` keeps one settled answer per (PR, question); the surrogate
|
|
45
|
+
-- `id` PK gives the `Table<T>` gateway a single-column key. Forward-only, additive (expand). Numbered
|
|
46
|
+
-- after the current highest committed prefix (104) in the pre-assigned 109–110 block (#806); the
|
|
47
|
+
-- runner wraps each file in its own transaction, so this file must NOT contain BEGIN/COMMIT.
|
|
48
|
+
CREATE TABLE IF NOT EXISTS pr_adjudications (
|
|
49
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
50
|
+
pr_key TEXT NOT NULL REFERENCES pull_requests(pr_key),
|
|
51
|
+
question_fingerprint TEXT NOT NULL,
|
|
52
|
+
answer TEXT,
|
|
53
|
+
adjudicated_by TEXT,
|
|
54
|
+
adjudicated_kind TEXT,
|
|
55
|
+
adjudicated_at TEXT NOT NULL,
|
|
56
|
+
invalidated_at TEXT,
|
|
57
|
+
source_completion_id INTEGER,
|
|
58
|
+
UNIQUE (pr_key, question_fingerprint)
|
|
59
|
+
);
|
|
60
|
+
CREATE INDEX IF NOT EXISTS idx_pradj_pr ON pr_adjudications(pr_key);
|
|
61
|
+
CREATE INDEX IF NOT EXISTS idx_pradj_srccompletion ON pr_adjudications(source_completion_id);
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
-- 110_task_completions_auto_applied.sql — issue #806 (Copilot review): mark an AUTO-APPLIED escalation
|
|
2
|
+
-- completion so it is distinguishable from a fresh human/agent submission in the attribution ledger.
|
|
3
|
+
--
|
|
4
|
+
-- The convergence poller auto-resumes an already-answered `wait-answer` by replaying a durable
|
|
5
|
+
-- adjudication through the SAME `completeUserTaskAttributed` door a human/agent uses (app/service.ts,
|
|
6
|
+
-- issue #806). Without a marker that replay is INDISTINGUISHABLE from a real, first-hand submission in
|
|
7
|
+
-- `task_completions`, and — recorded as an irreversible authority — it could launder an earlier
|
|
8
|
+
-- agent-originated decision into an unchallengeable human one. `auto_applied=1` records "this
|
|
9
|
+
-- completion is a machine replay of a prior decision, not a fresh submission"; the app also records
|
|
10
|
+
-- such completions `reversible=1` so a human can always override an auto-applied answer.
|
|
11
|
+
--
|
|
12
|
+
-- Forward-only, additive (expand): a nullable-defaulted `ADD COLUMN`, so every existing completion
|
|
13
|
+
-- reads back `auto_applied=0` (a genuine first-hand submission). Numbered after 109 in the pre-assigned
|
|
14
|
+
-- 109–110 block (#806); the runner wraps each file in its own transaction, so no BEGIN/COMMIT here.
|
|
15
|
+
ALTER TABLE task_completions ADD COLUMN auto_applied INTEGER NOT NULL DEFAULT 0;
|
|
16
|
+
|
|
17
|
+
-- Link an AUTO-APPLIED escalation completion back to the durable adjudication it replayed (issue #806,
|
|
18
|
+
-- Copilot review), so a human's revert of that completion can invalidate the exact decision. The
|
|
19
|
+
-- convergence poller auto-resumes an already-answered `wait-answer` by replaying a `pr_adjudications`
|
|
20
|
+
-- row through `completeEscalationAutoApplied` (app/service.ts); recording WHICH adjudication it replayed
|
|
21
|
+
-- lets `revertAgentCompletion` tombstone that decision (`invalidateAdjudication`) so the revert becomes a
|
|
22
|
+
-- real override — the next round re-parks a human — instead of the poller silently re-applying the same
|
|
23
|
+
-- overridden answer. NULL for every first-hand (human/agent) submission and for legacy rows; only an
|
|
24
|
+
-- auto-apply carries a source adjudication. Additive (expand): a nullable `ADD COLUMN`. Folded into this
|
|
25
|
+
-- file to keep the whole change inside the pre-assigned 109–110 block (#806, Copilot review) rather than
|
|
26
|
+
-- consuming an unallocated 111 prefix.
|
|
27
|
+
ALTER TABLE task_completions ADD COLUMN source_adjudication_id INTEGER;
|
|
28
|
+
|
|
29
|
+
-- `latestAdjudicator` (workers/answer-escalation) now looks a completion up by `process_instance_key`
|
|
30
|
+
-- for every convergence answer, to correlate the settled adjudicator's attribution (#806). The ledger
|
|
31
|
+
-- is append-only and only carried an index on `user_task_key` (026_agent_completion.sql), so that
|
|
32
|
+
-- lookup would scan the whole completion history as the fleet grows. Index `process_instance_key` too
|
|
33
|
+
-- (additive/expand — a new index, no existing shape touched).
|
|
34
|
+
CREATE INDEX idx_task_completions_pik ON task_completions(process_instance_key);
|
package/openapi.yaml
CHANGED
|
@@ -429,6 +429,7 @@ components:
|
|
|
429
429
|
- jobKeys
|
|
430
430
|
- live
|
|
431
431
|
- staleMs
|
|
432
|
+
- harnessStale
|
|
432
433
|
properties:
|
|
433
434
|
instance:
|
|
434
435
|
type: string
|
|
@@ -456,6 +457,17 @@ components:
|
|
|
456
457
|
staleMs:
|
|
457
458
|
type: integer
|
|
458
459
|
description: Milliseconds since the last liveness refresh (0 when fresh).
|
|
460
|
+
harnessProtocol:
|
|
461
|
+
type: integer
|
|
462
|
+
minimum: 0
|
|
463
|
+
description: The worker-harness protocol version this worker advertised at enrolment (issue #802), when a numeric one is known.
|
|
464
|
+
harnessStale:
|
|
465
|
+
type: boolean
|
|
466
|
+
description: >-
|
|
467
|
+
Whether this worker's harness is STALE (issue #802) — below the configured minimum protocol
|
|
468
|
+
or advertising no version at all — so it may silently swallow AgentInstance / transcript /
|
|
469
|
+
result-envelope artifacts. Surfaced so the operator can drain it. Distinct from the
|
|
470
|
+
liveness `staleMs` heartbeat grade.
|
|
459
471
|
AgenticSupplyLeaf:
|
|
460
472
|
type: object
|
|
461
473
|
description: The supply for one leaf token — the workers registered under it.
|
|
@@ -622,6 +634,16 @@ components:
|
|
|
622
634
|
sets false) redrives a re-leased round from scratch. Recorded only when `instance` is
|
|
623
635
|
a non-blank string — a missing, empty, or whitespace-only `instance` is echoed back for
|
|
624
636
|
provenance but the flag is not persisted.
|
|
637
|
+
harnessProtocol:
|
|
638
|
+
type: integer
|
|
639
|
+
minimum: 0
|
|
640
|
+
description: >-
|
|
641
|
+
The worker-harness protocol version (issue #802) — a non-negative integer declaring which
|
|
642
|
+
machine-readable artifacts the harness emits (AgentInstance, transcript flush, result
|
|
643
|
+
envelope). An ENROLMENT attribute, never a routing token. Recorded per instance so the app
|
|
644
|
+
can flag a stale harness in getAgenticSupply / the registry and — under
|
|
645
|
+
NANO_AGENTIC_STALE_HARNESS_POLICY=refuse — refuse it agent-job routing. A missing version is
|
|
646
|
+
treated as stale.
|
|
625
647
|
EnrolledRole:
|
|
626
648
|
type: object
|
|
627
649
|
description: One matched role in an enrolment resolution — provenance for the resolved SERVE set.
|
|
@@ -643,6 +665,7 @@ components:
|
|
|
643
665
|
- roles
|
|
644
666
|
- demandVersion
|
|
645
667
|
- leaseTtl
|
|
668
|
+
- harnessStale
|
|
646
669
|
properties:
|
|
647
670
|
instance:
|
|
648
671
|
type: string
|
|
@@ -655,6 +678,20 @@ components:
|
|
|
655
678
|
guarantee of durable persistence — recording into the durable-resume registry is
|
|
656
679
|
best-effort (skipped when `instance` is absent/blank, and a registry write hiccup is
|
|
657
680
|
logged without failing enrolment).
|
|
681
|
+
harnessProtocol:
|
|
682
|
+
type: integer
|
|
683
|
+
minimum: 0
|
|
684
|
+
description: >-
|
|
685
|
+
Echo of the request's advertised harness protocol version (issue #802). Present only when
|
|
686
|
+
the request supplied it.
|
|
687
|
+
harnessStale:
|
|
688
|
+
type: boolean
|
|
689
|
+
description: >-
|
|
690
|
+
Whether this worker's harness is STALE (issue #802) — below the configured minimum protocol
|
|
691
|
+
(NANO_AGENTIC_MIN_HARNESS_PROTOCOL) or advertising no version at all. Always present. Under
|
|
692
|
+
NANO_AGENTIC_STALE_HARNESS_POLICY=refuse a stale harness is handed an EMPTY `serve` set so
|
|
693
|
+
it wins no job leases; under the default `flag` policy `serve` is unchanged and the worker
|
|
694
|
+
is only flagged for observability/drain.
|
|
658
695
|
serve:
|
|
659
696
|
type: array
|
|
660
697
|
description: The SERVE token set — sorted, de-duplicated leaf tokens the worker may serve.
|
|
@@ -812,7 +849,36 @@ components:
|
|
|
812
849
|
status:
|
|
813
850
|
type: string
|
|
814
851
|
enum: [green, amber, red]
|
|
815
|
-
description:
|
|
852
|
+
description: >-
|
|
853
|
+
The overall SLO — worst of the missing-agent signal and the diversity SLO, and folded to
|
|
854
|
+
`red` when any enrolled harness is stale (`staleWorkers` non-empty, issue #802), since the
|
|
855
|
+
board renders only this pill as its overall signal.
|
|
856
|
+
staleWorkers:
|
|
857
|
+
type: array
|
|
858
|
+
description: >-
|
|
859
|
+
The enrolled workers whose harness is STALE (issue #802) — below the configured minimum
|
|
860
|
+
protocol or advertising no version at all — so they may silently swallow AgentInstance /
|
|
861
|
+
transcript / result-envelope artifacts and should be drained. Present (possibly empty) when
|
|
862
|
+
the app's harness-protocol registry is available.
|
|
863
|
+
items:
|
|
864
|
+
$ref: "#/components/schemas/StaleWorker"
|
|
865
|
+
StaleWorker:
|
|
866
|
+
type: object
|
|
867
|
+
description: One enrolled worker flagged as running a stale harness (issue #802).
|
|
868
|
+
required:
|
|
869
|
+
- instance
|
|
870
|
+
- stale
|
|
871
|
+
properties:
|
|
872
|
+
instance:
|
|
873
|
+
type: string
|
|
874
|
+
description: The worker instance id.
|
|
875
|
+
harnessProtocol:
|
|
876
|
+
type: integer
|
|
877
|
+
minimum: 0
|
|
878
|
+
description: The advertised harness protocol version, when a numeric one is known (omitted when none was advertised).
|
|
879
|
+
stale:
|
|
880
|
+
type: boolean
|
|
881
|
+
description: Whether the worker's harness is stale (always true for entries in this list).
|
|
816
882
|
AgenticTranscript:
|
|
817
883
|
type: object
|
|
818
884
|
description: One captured agent session's transcript metadata (H3/#146 transcript store). A durable
|
|
@@ -3993,6 +4059,10 @@ paths:
|
|
|
3993
4059
|
durableResume:
|
|
3994
4060
|
type: boolean
|
|
3995
4061
|
description: "Whether this worker's harness advertises durable-resume (issue #325, ADR 0062 Slice 5/5) — an ENROLMENT attribute, never a routing token. Recorded per instance so the app emits the world-restore marker only to a fleet with a participant; a harness that omits it (or sets false) redrives a re-leased round from scratch. Recorded only when `instance` is a non-blank string — a missing, empty, or whitespace-only `instance` is echoed back for provenance but the flag is not persisted."
|
|
4062
|
+
harnessProtocol:
|
|
4063
|
+
type: integer
|
|
4064
|
+
minimum: 0
|
|
4065
|
+
description: 'The worker-harness protocol version (issue #802) — a non-negative integer declaring which machine-readable artifacts the harness emits (AgentInstance, transcript flush, result envelope). An ENROLMENT attribute, never a routing token. Recorded per instance so the app can flag a stale harness in getAgenticSupply / the registry and — under NANO_AGENTIC_STALE_HARNESS_POLICY=refuse — refuse it agent-job routing. A missing version is treated as stale.'
|
|
3996
4066
|
# END generated:mcp-body
|
|
3997
4067
|
responses:
|
|
3998
4068
|
"200":
|
|
@@ -80,7 +80,7 @@ test("complete-user-task: completes a plan-review escalation and drops its read-
|
|
|
80
80
|
assertEquals(res.status, 200);
|
|
81
81
|
assertEquals(res.body.ok, true);
|
|
82
82
|
assertEquals(res.body.elementId, "plan-review-decision");
|
|
83
|
-
assertEquals(completed, [{ userTaskKey: "ut-1", variables: { directive: "revise", notes: "narrow scope" } }]);
|
|
83
|
+
assertEquals(completed, [{ userTaskKey: "ut-1", variables: { directive: "revise", notes: "narrow scope", completedUserTaskKey: "ut-1", completedCompletionId: 1 } }]);
|
|
84
84
|
assertEquals(stores.user_tasks, []);
|
|
85
85
|
// Attribution recorded as a human completion.
|
|
86
86
|
assertEquals(stores.task_completions.length, 1);
|
|
@@ -93,7 +93,7 @@ test("complete-user-task: completes a trial-merge escalation with the typed acti
|
|
|
93
93
|
const res = await call(app, { userTaskKey: "ut-2", variables: { action: "rebase" } });
|
|
94
94
|
|
|
95
95
|
assertEquals(res.status, 200);
|
|
96
|
-
assertEquals(completed, [{ userTaskKey: "ut-2", variables: { action: "rebase" } }]);
|
|
96
|
+
assertEquals(completed, [{ userTaskKey: "ut-2", variables: { action: "rebase", completedUserTaskKey: "ut-2", completedCompletionId: 1 } }]);
|
|
97
97
|
});
|
|
98
98
|
|
|
99
99
|
test("complete-user-task: a missing userTaskKey is a 400", async () => {
|
|
@@ -120,7 +120,7 @@ test("complete-user-task: completes a feature-blocked acknowledgement with the t
|
|
|
120
120
|
const res = await call(app, { userTaskKey: "ut-4", variables: { note: "reassigned" } });
|
|
121
121
|
assertEquals(res.status, 200);
|
|
122
122
|
assertEquals(res.body.elementId, "feature-blocked");
|
|
123
|
-
assertEquals(completed, [{ userTaskKey: "ut-4", variables: { note: "reassigned" } }]);
|
|
123
|
+
assertEquals(completed, [{ userTaskKey: "ut-4", variables: { note: "reassigned", completedUserTaskKey: "ut-4", completedCompletionId: 1 } }]);
|
|
124
124
|
});
|
|
125
125
|
|
|
126
126
|
test("complete-user-task: completes a feature-escalation answer (issue #332)", async () => {
|
|
@@ -128,7 +128,7 @@ test("complete-user-task: completes a feature-escalation answer (issue #332)", a
|
|
|
128
128
|
const res = await call(app, { userTaskKey: "ut-6", variables: { resolution: "answer", answer: "use v2" } });
|
|
129
129
|
assertEquals(res.status, 200);
|
|
130
130
|
assertEquals(res.body.elementId, "feature-escalation");
|
|
131
|
-
assertEquals(completed, [{ userTaskKey: "ut-6", variables: { resolution: "answer", answer: "use v2" } }]);
|
|
131
|
+
assertEquals(completed, [{ userTaskKey: "ut-6", variables: { resolution: "answer", answer: "use v2", completedUserTaskKey: "ut-6", completedCompletionId: 1 } }]);
|
|
132
132
|
});
|
|
133
133
|
|
|
134
134
|
test("complete-user-task: refuses a non-completable internal user task (400)", async () => {
|
|
@@ -161,5 +161,5 @@ test("complete-user-task: a read-model cleanup failure does not mask a resumed c
|
|
|
161
161
|
|
|
162
162
|
assertEquals(res.status, 200);
|
|
163
163
|
assertEquals(res.body.ok, true);
|
|
164
|
-
assertEquals(completed, [{ userTaskKey: "ut-5", variables: { directive: "revise", notes: "narrow scope" } }]);
|
|
164
|
+
assertEquals(completed, [{ userTaskKey: "ut-5", variables: { directive: "revise", notes: "narrow scope", completedUserTaskKey: "ut-5", completedCompletionId: 1 } }]);
|
|
165
165
|
});
|
|
@@ -4,6 +4,7 @@ import { assert, assertEquals } from "#test-assert";
|
|
|
4
4
|
import type { AppApi } from "@nanobpm/urban";
|
|
5
5
|
import { memDataFor } from "../test/worldDb.ts";
|
|
6
6
|
import { DurableResumeRegistry } from "../app/durableResume.ts";
|
|
7
|
+
import { HarnessProtocolRegistry } from "../app/harnessProtocol.ts";
|
|
7
8
|
import { noopLog } from "../test/log.ts";
|
|
8
9
|
import handler from "./enrolAgenticWorker.ts";
|
|
9
10
|
|
|
@@ -151,3 +152,86 @@ test("enforces the shared secret when NANO_PR_WEBHOOK_SECRET is set", async () =
|
|
|
151
152
|
else process.env["NANO_PR_WEBHOOK_SECRET"] = prev;
|
|
152
153
|
}
|
|
153
154
|
});
|
|
155
|
+
|
|
156
|
+
// Harness-protocol enrolment gate (issue #802).
|
|
157
|
+
const HARNESS_MIGRATIONS = ["052_worker_durable_resume.sql", "107_worker_harness_protocol.sql"];
|
|
158
|
+
|
|
159
|
+
test("echoes harnessProtocol and reports harnessStale=false for a healthy protocol (>= minimum)", async () => {
|
|
160
|
+
const res = (await handler(input({ capability: { cognition: "decide" }, instance: "w1", harnessProtocol: 2 }), app)) as any;
|
|
161
|
+
assertEquals(res.status, 200);
|
|
162
|
+
assertEquals(res.body.harnessProtocol, 2);
|
|
163
|
+
assertEquals(res.body.harnessStale, false);
|
|
164
|
+
// No routing regression under the default `flag` policy: SERVE is unchanged.
|
|
165
|
+
assert(res.body.serve.includes("decide"));
|
|
166
|
+
});
|
|
167
|
+
|
|
168
|
+
test("flags harnessStale=true when the harness advertises no version at all (absent = stale)", async () => {
|
|
169
|
+
const res = (await handler(input({ capability: { cognition: "decide" }, instance: "w1" }), app)) as any;
|
|
170
|
+
assertEquals(res.status, 200);
|
|
171
|
+
assertEquals("harnessProtocol" in res.body, false, "no protocol echoed when none advertised");
|
|
172
|
+
assertEquals(res.body.harnessStale, true);
|
|
173
|
+
// Default `flag` policy: a stale harness is still routed (only flagged), so no fleet regression.
|
|
174
|
+
assert(res.body.serve.includes("decide"), "flag policy leaves SERVE intact");
|
|
175
|
+
});
|
|
176
|
+
|
|
177
|
+
test("flags harnessStale=true for a below-minimum protocol", async () => {
|
|
178
|
+
const prev = process.env["NANO_AGENTIC_MIN_HARNESS_PROTOCOL"];
|
|
179
|
+
process.env["NANO_AGENTIC_MIN_HARNESS_PROTOCOL"] = "3";
|
|
180
|
+
try {
|
|
181
|
+
const res = (await handler(input({ capability: { cognition: "decide" }, instance: "w1", harnessProtocol: 1 }), app)) as any;
|
|
182
|
+
assertEquals(res.status, 200);
|
|
183
|
+
assertEquals(res.body.harnessStale, true);
|
|
184
|
+
} finally {
|
|
185
|
+
if (prev === undefined) delete process.env["NANO_AGENTIC_MIN_HARNESS_PROTOCOL"];
|
|
186
|
+
else process.env["NANO_AGENTIC_MIN_HARNESS_PROTOCOL"] = prev;
|
|
187
|
+
}
|
|
188
|
+
});
|
|
189
|
+
|
|
190
|
+
test("rejects a non-integer/negative harnessProtocol as 400", async () => {
|
|
191
|
+
const nonInt = (await handler(input({ capability: { cognition: "decide" }, harnessProtocol: 1.5 }), app)) as any;
|
|
192
|
+
assertEquals(nonInt.status, 400);
|
|
193
|
+
const negative = (await handler(input({ capability: { cognition: "decide" }, harnessProtocol: -1 }), app)) as any;
|
|
194
|
+
assertEquals(negative.status, 400);
|
|
195
|
+
const str = (await handler(input({ capability: { cognition: "decide" }, harnessProtocol: "2" }), app)) as any;
|
|
196
|
+
assertEquals(str.status, 400);
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
test("records the advertised harness protocol in the registry when a data layer + instance are present", async () => {
|
|
200
|
+
const { data } = memDataFor(HARNESS_MIGRATIONS);
|
|
201
|
+
const withData = { log: noopLog(), data } as unknown as AppApi;
|
|
202
|
+
const res = (await handler(input({ capability: { cognition: "decide" }, instance: "w1", harnessProtocol: 2 }), withData)) as any;
|
|
203
|
+
assertEquals(res.status, 200);
|
|
204
|
+
assertEquals(await new HarnessProtocolRegistry(data).protocolFor("w1"), 2);
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
test("a re-enrol WITHOUT a protocol clears a stale-healthy recorded value (degrade to stale)", async () => {
|
|
208
|
+
const { data } = memDataFor(HARNESS_MIGRATIONS);
|
|
209
|
+
const withData = { log: noopLog(), data } as unknown as AppApi;
|
|
210
|
+
await handler(input({ capability: { cognition: "decide" }, instance: "w1", harnessProtocol: 3 }), withData);
|
|
211
|
+
assertEquals(await new HarnessProtocolRegistry(data).protocolFor("w1"), 3);
|
|
212
|
+
const res = (await handler(input({ capability: { cognition: "decide" }, instance: "w1" }), withData)) as any;
|
|
213
|
+
assertEquals(res.status, 200);
|
|
214
|
+
assertEquals(await new HarnessProtocolRegistry(data).protocolFor("w1"), undefined, "stale-healthy value cleared");
|
|
215
|
+
});
|
|
216
|
+
|
|
217
|
+
test("under the `refuse` policy a stale harness is handed an EMPTY SERVE set (no job leases)", async () => {
|
|
218
|
+
const prev = process.env["NANO_AGENTIC_STALE_HARNESS_POLICY"];
|
|
219
|
+
process.env["NANO_AGENTIC_STALE_HARNESS_POLICY"] = "refuse";
|
|
220
|
+
try {
|
|
221
|
+
const mod = await import(`./enrolAgenticWorker.ts?refuse=${Date.now()}`);
|
|
222
|
+
const guarded = mod.default as typeof handler;
|
|
223
|
+
// A stale (version-less) worker: SERVE withheld.
|
|
224
|
+
const stale = (await guarded(input({ capability: { cognition: "decide" }, instance: "w1" }), app)) as any;
|
|
225
|
+
assertEquals(stale.status, 200);
|
|
226
|
+
assertEquals(stale.body.harnessStale, true);
|
|
227
|
+
assertEquals(stale.body.serve, [], "refuse policy withholds SERVE for a stale harness");
|
|
228
|
+
assertEquals(stale.body.roles, []);
|
|
229
|
+
// A healthy worker is routed exactly as today.
|
|
230
|
+
const healthy = (await guarded(input({ capability: { cognition: "decide" }, instance: "w2", harnessProtocol: 5 }), app)) as any;
|
|
231
|
+
assertEquals(healthy.body.harnessStale, false);
|
|
232
|
+
assert(healthy.body.serve.includes("decide"), "healthy harness routed under refuse policy");
|
|
233
|
+
} finally {
|
|
234
|
+
if (prev === undefined) delete process.env["NANO_AGENTIC_STALE_HARNESS_POLICY"];
|
|
235
|
+
else process.env["NANO_AGENTIC_STALE_HARNESS_POLICY"] = prev;
|
|
236
|
+
}
|
|
237
|
+
});
|