@openwop/openwop-conformance 2.40.3 → 2.41.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,198 @@
1
+ /**
2
+ * RFC 0111 — the transcript-window checker, kept pure so its sabotage cases are
3
+ * self-tests (`context-budget.test.ts`) rather than claims.
4
+ *
5
+ * WHY THIS FILE EXISTS (the 2026-09-26 correction on record, COMPATIBILITY.md).
6
+ * Until suite 2.41.0 `context-budget-transcript-bound` could not fail in the ways
7
+ * RFC 0111 names:
8
+ * (b) "the harness independently … re-computes their token sum" — impossible:
9
+ * events are content-free, so the suite had nothing to count. The test
10
+ * checked that each id existed, and nothing else.
11
+ * (c) "the recent tail" — the test checked only that ids were unique.
12
+ * and no check that the budget was ever under PRESSURE, so a host advertising
13
+ * a budget of 10^9 passed.
14
+ * The seam now also returns `entries[] { eventId, rendered }` — the exact text
15
+ * the host fed the model for each event that iteration — and for
16
+ * `tokenCounter: "chars"` the suite recounts it itself.
17
+ *
18
+ * Definitions this checker applies (host-sample-test-seams.md §14):
19
+ * - `chars` is Unicode code points (`[...s].length`), not UTF-16 units or bytes.
20
+ * - An event is ELIGIBLE for an iteration's transcript when it shares
21
+ * `(type, nodeId)` with an event the host fed verbatim that iteration. The
22
+ * RFC's "no older event included while a newer eligible one is dropped"
23
+ * never defined "eligible"; keying on the pair (not the type alone) keeps a
24
+ * host that feeds only one node's events from being convicted for another
25
+ * node's.
26
+ * - PRESSURE: the budget did something — an eligible event older than the
27
+ * oldest fed one exists, or the window carries a summarized range.
28
+ */
29
+
30
+ export interface WindowEntry {
31
+ readonly eventId: string;
32
+ readonly rendered: string;
33
+ }
34
+ export interface SummarizedRange {
35
+ readonly summaryRef: string;
36
+ readonly replacedTurns: readonly string[];
37
+ }
38
+ export interface TranscriptWindow {
39
+ readonly tokenCounter: string;
40
+ readonly tokenCount: number;
41
+ readonly eventIds: readonly string[];
42
+ readonly summarizedRanges: readonly SummarizedRange[];
43
+ readonly entries?: readonly WindowEntry[];
44
+ }
45
+ export interface LogEvent {
46
+ readonly eventId: string;
47
+ readonly type: string;
48
+ readonly sequence: number;
49
+ readonly nodeId?: string;
50
+ readonly payload: Record<string, unknown>;
51
+ }
52
+ export interface BudgetCap {
53
+ readonly transcriptTokenBudget: number;
54
+ readonly tokenCounter: string;
55
+ }
56
+ export interface WindowVerdict {
57
+ /** Every broken rule, as a sentence naming the iteration. Empty = consistent. */
58
+ readonly violations: string[];
59
+ /** Whether this iteration shows the budget acting (see file docblock). */
60
+ readonly pressure: boolean;
61
+ /** Rules that could not be evaluated (no event log). */
62
+ readonly unmeasured: string[];
63
+ }
64
+
65
+ const isRecord = (v: unknown): v is Record<string, unknown> => typeof v === 'object' && v !== null && !Array.isArray(v);
66
+ const stringArrayOf = (v: unknown): string[] | undefined => (Array.isArray(v) && v.every((x) => typeof x === 'string') ? v : undefined);
67
+
68
+ /** Unicode code points — the `chars` unit. */
69
+ export function charCount(s: string): number {
70
+ return [...s].length;
71
+ }
72
+
73
+ /** Parse a seam response; undefined when the required shape is wrong (a malformed optional member is also undefined — it cannot be half-trusted). */
74
+ export function parseTranscriptWindow(v: unknown): TranscriptWindow | undefined {
75
+ if (!isRecord(v)) return undefined;
76
+ const tokenCounter = v['tokenCounter'];
77
+ const tokenCount = v['tokenCount'];
78
+ const eventIds = stringArrayOf(v['eventIds']);
79
+ if (typeof tokenCounter !== 'string' || typeof tokenCount !== 'number' || eventIds === undefined) return undefined;
80
+ const summarizedRanges: SummarizedRange[] = [];
81
+ const rawRanges = v['summarizedRanges'];
82
+ if (rawRanges !== undefined) {
83
+ if (!Array.isArray(rawRanges)) return undefined;
84
+ for (const r of rawRanges) {
85
+ if (!isRecord(r) || typeof r['summaryRef'] !== 'string') return undefined;
86
+ const replacedTurns = stringArrayOf(r['replacedTurns']);
87
+ if (replacedTurns === undefined) return undefined;
88
+ summarizedRanges.push({ summaryRef: r['summaryRef'], replacedTurns });
89
+ }
90
+ }
91
+ const rawEntries = v['entries'];
92
+ if (rawEntries === undefined) return { tokenCounter, tokenCount, eventIds, summarizedRanges };
93
+ if (!Array.isArray(rawEntries)) return undefined;
94
+ const entries: WindowEntry[] = [];
95
+ for (const e of rawEntries) {
96
+ if (!isRecord(e) || typeof e['eventId'] !== 'string' || typeof e['rendered'] !== 'string') return undefined;
97
+ entries.push({ eventId: e['eventId'], rendered: e['rendered'] });
98
+ }
99
+ return { tokenCounter, tokenCount, eventIds, summarizedRanges, entries };
100
+ }
101
+
102
+ const eligibilityKey = (e: LogEvent): string => `${e.type}\u0000${e.nodeId ?? ''}`;
103
+
104
+ /**
105
+ * Check one iteration's window against the advertised cap and (when available)
106
+ * the run's event log. `log` null ⇒ the log-dependent rules are reported in
107
+ * `unmeasured`, never silently passed.
108
+ */
109
+ export function checkTranscriptWindow(iteration: number, w: TranscriptWindow, cap: BudgetCap, log: readonly LogEvent[] | null): WindowVerdict {
110
+ const violations: string[] = [];
111
+ const unmeasured: string[] = [];
112
+ const at = `iteration ${iteration}`;
113
+
114
+ if (w.tokenCounter !== cap.tokenCounter) violations.push(`${at}: seam tokenCounter "${w.tokenCounter}" is not the advertised "${cap.tokenCounter}"`);
115
+ if (w.tokenCount > cap.transcriptTokenBudget) violations.push(`${at}: tokenCount ${w.tokenCount} exceeds transcriptTokenBudget ${cap.transcriptTokenBudget}`);
116
+ if (new Set(w.eventIds).size !== w.eventIds.length) violations.push(`${at}: eventIds repeats an entry`);
117
+
118
+ // (b) — the recount. Only `chars` is recountable by a black-box harness.
119
+ if (cap.tokenCounter === 'chars') {
120
+ if (w.entries === undefined) {
121
+ violations.push(`${at}: tokenCounter is "chars" but the seam returned no entries[] — the recount (RFC 0111 (b)) cannot run, and for "chars" entries are REQUIRED`);
122
+ } else {
123
+ const sum = w.entries.reduce((n, e) => n + charCount(e.rendered), 0);
124
+ if (sum !== w.tokenCount) violations.push(`${at}: entries[].rendered sum to ${sum} chars but tokenCount is ${w.tokenCount}`);
125
+ }
126
+ }
127
+
128
+ if (log === null) {
129
+ unmeasured.push('real-event, entries↔eventIds, summary-range, recent-tail and pressure rules (no event log)');
130
+ return { violations, pressure: false, unmeasured };
131
+ }
132
+
133
+ const byId = new Map(log.map((e) => [e.eventId, e]));
134
+ const summarizedByRef = new Map<string, LogEvent>();
135
+ for (const e of log) {
136
+ if (e.type !== 'context.summarized') continue;
137
+ const ref = e.payload['summaryRef'];
138
+ if (typeof ref === 'string') summarizedByRef.set(ref, e);
139
+ }
140
+
141
+ for (const id of w.eventIds) if (!byId.has(id)) violations.push(`${at}: eventId "${id}" is not an event of this run`);
142
+ for (const r of w.summarizedRanges) if (!summarizedByRef.has(r.summaryRef)) violations.push(`${at}: summarizedRanges summaryRef "${r.summaryRef}" has no matching context.summarized event`);
143
+
144
+ if (w.entries !== undefined) {
145
+ const verbatim: string[] = [];
146
+ for (const e of w.entries) {
147
+ const ev = byId.get(e.eventId);
148
+ if (ev === undefined) { violations.push(`${at}: entries[] eventId "${e.eventId}" is not an event of this run`); continue; }
149
+ if (ev.type === 'context.summarized') {
150
+ const ref = ev.payload['summaryRef'];
151
+ if (!w.summarizedRanges.some((r) => r.summaryRef === ref)) violations.push(`${at}: summary entry "${e.eventId}" is not listed in summarizedRanges`);
152
+ } else {
153
+ verbatim.push(e.eventId);
154
+ }
155
+ }
156
+ if (verbatim.join('\u0000') !== w.eventIds.join('\u0000')) violations.push(`${at}: the verbatim entries[] (${verbatim.length}) are not eventIds (${w.eventIds.length}) in the same order`);
157
+ }
158
+
159
+ // (c) — the recent tail.
160
+ const fed = w.eventIds.map((id) => byId.get(id)).filter((e): e is LogEvent => e !== undefined);
161
+ if (fed.length === 0) return { violations, pressure: w.summarizedRanges.length > 0, unmeasured };
162
+ for (let i = 1; i < fed.length; i += 1) {
163
+ if (fed[i].sequence <= fed[i - 1].sequence) { violations.push(`${at}: eventIds are not in event-log order`); break; }
164
+ }
165
+ const decided = log.find((e) => e.type === 'runOrchestrator.decided' && e.payload['iteration'] === iteration);
166
+ const cutoff = decided?.sequence ?? Number.POSITIVE_INFINITY;
167
+ for (const e of fed) if (e.sequence >= cutoff) violations.push(`${at}: eventId "${e.eventId}" (seq ${e.sequence}) is at or after the iteration's own decision (seq ${cutoff})`);
168
+
169
+ const keys = new Set(fed.map(eligibilityKey));
170
+ const fedIds = new Set(fed.map((e) => e.eventId));
171
+ const replaced = new Set(w.summarizedRanges.flatMap((r) => r.replacedTurns));
172
+ const minFed = Math.min(...fed.map((e) => e.sequence));
173
+ const maxBound = Number.isFinite(cutoff) ? cutoff : Math.max(...fed.map((e) => e.sequence)) + 1;
174
+ let pressure = w.summarizedRanges.length > 0;
175
+ for (const e of log) {
176
+ if (!keys.has(eligibilityKey(e)) || fedIds.has(e.eventId)) continue;
177
+ if (e.sequence > minFed && e.sequence < maxBound && !replaced.has(e.eventId)) {
178
+ violations.push(`${at}: eligible event "${e.eventId}" (${e.type}, seq ${e.sequence}) was dropped while an older one (seq ${minFed}) was fed — not a recent tail`);
179
+ }
180
+ if (e.sequence < minFed) pressure = true;
181
+ }
182
+ return { violations, pressure, unmeasured };
183
+ }
184
+
185
+ /** The summary texts a window fed, in entry order — the model-facing proof of replay reuse. */
186
+ export function summaryTexts(w: TranscriptWindow, log: readonly LogEvent[]): string[] {
187
+ if (w.entries === undefined) return [];
188
+ const summaryIds = new Set(log.filter((e) => e.type === 'context.summarized').map((e) => e.eventId));
189
+ return w.entries.filter((e) => summaryIds.has(e.eventId)).map((e) => e.rendered);
190
+ }
191
+
192
+ /** v1 reads `summarization.supported: true`; v2 has no `supported` seat — presence of the record is the claim (capabilities.md §2). */
193
+ export function summarizationDeclared(contextBudget: unknown, major: 1 | 2): boolean {
194
+ if (!isRecord(contextBudget)) return false;
195
+ const s = contextBudget['summarization'];
196
+ if (!isRecord(s)) return false;
197
+ return major === 2 ? true : s['supported'] === true;
198
+ }
@@ -74,9 +74,17 @@
74
74
  */
75
75
  export const HARNESS_DOUBLE_MODULES: readonly string[] = [
76
76
  'a2a-fake-peer',
77
+ // The two receiver modules whose only job is a server the HOST calls (2.40.4).
78
+ // They were missing, so `v2-a2a-push-delivery`'s correct declaration read as a
79
+ // callback it "does not make" and Conformance Soak went red on every run from
80
+ // #1546 (2026-09-24) on, while eight webhook/effect scenarios that DO take a
81
+ // host callback declared nothing. `webhook-receiver` stays off the list: it is
82
+ // mostly verification helpers, and importing it proves nothing.
83
+ 'effect-receiver',
77
84
  'mcp-fake-server',
78
85
  'oidc-issuer',
79
86
  'otel-collector',
87
+ 'scoped-receiver',
80
88
  ];
81
89
 
82
90
  /**
@@ -16,18 +16,32 @@
16
16
  * summarizedRanges }`. The seam is OPTIONAL — the scenario soft-skips on
17
17
  * `404`/`405` (the RFC defers reference-host implementation).
18
18
  *
19
- * Asserts, for each iteration the host reports:
19
+ * Asserts, for each iteration the host reports (the rules live in the pure
20
+ * checker `lib/context-budget.ts`, whose sabotage cases are self-tests):
20
21
  * 1. `tokenCounter` equals the advertised `contextBudget.tokenCounter`.
21
22
  * 2. `tokenCount ≤ transcriptTokenBudget` (the per-turn bound).
22
- * 3. CROSS-CHECK — the harness independently reads the events named in
23
- * `eventIds` from the run event-log (`/v1/host/sample/test/runs/:runId/events`)
24
- * and confirms every named id is a real persisted event of the run, so the
25
- * host's reported accounting is internally consistent (not fabricated).
26
- * 4. RECENT-TAIL — `eventIds` are a contiguous most-recent suffix of the run's
27
- * eligible event-log entries (no older event included while a newer eligible
28
- * one is dropped).
29
- * 5. SUMMARIZED-RANGE — every `summarizedRanges[].summaryRef` has a matching
30
- * `context.summarized` event in the run event-log.
23
+ * 3. RECOUNT (RFC 0111 (b), corrected 2026-09-26) — for `tokenCounter:
24
+ * "chars"` the seam MUST return `entries[] { eventId, rendered }` and the
25
+ * suite sums the rendered text's code points itself; the sum MUST equal
26
+ * `tokenCount`. Other units are advertise-and-attest.
27
+ * 4. REAL EVENTS — every `eventIds` / `entries[]` id is an event of the run;
28
+ * the verbatim entries are `eventIds` in order.
29
+ * 5. RECENT TAIL — fed events are in log order, none at or after the
30
+ * iteration's own `runOrchestrator.decided`, and no eligible event (same
31
+ * `(type, nodeId)`) newer than the oldest fed one was dropped.
32
+ * 6. SUMMARIZED-RANGE — every `summaryRef` has a `context.summarized` event.
33
+ * 7. PRESSURE — at least one iteration shows the budget acting (an older
34
+ * eligible event evicted, or a summarized range). Without it the row
35
+ * records `partial-witness:` — a budget of 10^9 is not a witness.
36
+ *
37
+ * Both majors (2.41.0). Major 1 gates through `behaviorGate`; major 2 reads the
38
+ * `multiAgent` record's presence and records `inapplicable` when
39
+ * `contextBudget` is absent — never a strict-mode failure for a host that does
40
+ * not claim the capability. The seam is in the seams profile at major 2.
41
+ * Gated on the LIVE fixture `conformance-context-budget-live` (no mock
42
+ * decisions): the scripted `conformance-context-budget-multiturn` drives the
43
+ * supervisor through `mockDecisions`, which is exactly the mock RFC 0111
44
+ * §Scope forbids from advertising `contextBudget`.
31
45
  *
32
46
  * Honest non-vacuity ceiling (RFC 0111 §"Conformance seam"): the model-facing
33
47
  * prompt is genuinely host-internal, so this proves the host's DECLARED
@@ -47,209 +61,97 @@ import { isFixtureAdvertised } from '../lib/fixtures.js';
47
61
  import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
48
62
  import { queryTestEvents } from '../lib/event-log-query.js';
49
63
  import { req } from '../lib/requirement-ids.js';
50
- import { softSkip } from '../lib/soft-skip.js';
64
+ import { softSkip, seamAbsent } from '../lib/soft-skip.js';
65
+ import { seamsProfileAdvertised, targetMajor } from '../lib/seams.js';
66
+ import { familyAdvertised, v2Discovery } from '../lib/v2.js';
67
+ import { runsPath } from '../lib/memoryAttribution.js';
68
+ import { checkTranscriptWindow, parseTranscriptWindow, type LogEvent, type TranscriptWindow } from '../lib/context-budget.js';
51
69
 
52
- const FIXTURE = 'conformance-context-budget-multiturn';
70
+ const FIXTURE = 'conformance-context-budget-live';
53
71
  const PROFILE = 'openwop-context-budget';
54
72
  const MAX_ITERATIONS_PROBED = 16;
73
+ const ID = 'openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi';
55
74
 
56
- interface SummarizationCap {
57
- readonly supported?: boolean;
58
- readonly strategy?: string;
59
- readonly keepLastTurns?: number;
60
- }
61
- interface ContextBudgetCap {
62
- readonly transcriptTokenBudget?: number;
63
- readonly tokenCounter?: string;
64
- readonly summarization?: SummarizationCap;
65
- }
66
- interface ExecutionModelCap {
67
- readonly contextBudget?: ContextBudgetCap;
68
- }
69
- interface MultiAgentCap {
70
- readonly executionModel?: ExecutionModelCap;
71
- }
72
-
73
- // ── cast-free typed accessors (no `as`) ──────────────────────────────────
74
75
  function isRecord(v: unknown): v is Record<string, unknown> {
75
76
  return typeof v === 'object' && v !== null && !Array.isArray(v);
76
77
  }
77
- function isString(v: unknown): v is string {
78
- return typeof v === 'string';
79
- }
80
- function isNumber(v: unknown): v is number {
81
- return typeof v === 'number';
82
- }
83
- function stringOf(v: unknown): string | undefined {
84
- return isString(v) ? v : undefined;
85
- }
86
- function numberOf(v: unknown): number | undefined {
87
- return isNumber(v) ? v : undefined;
88
- }
89
- function stringArrayOf(v: unknown): string[] | undefined {
90
- return Array.isArray(v) && v.every(isString) ? v : undefined;
78
+ function recordAt(v: unknown, ...keys: string[]): Record<string, unknown> | undefined {
79
+ let cur: unknown = v;
80
+ for (const k of keys) cur = isRecord(cur) ? cur[k] : undefined;
81
+ return isRecord(cur) ? cur : undefined;
91
82
  }
92
83
  function runIdOf(v: unknown): string | undefined {
93
- return isRecord(v) ? stringOf(v['runId']) : undefined;
94
- }
95
-
96
- interface SummarizedRange {
97
- readonly summaryRef: string;
98
- readonly replacedTurns: string[];
99
- }
100
- interface TranscriptWindow {
101
- readonly tokenCounter: string;
102
- readonly tokenCount: number;
103
- readonly eventIds: string[];
104
- readonly summarizedRanges: SummarizedRange[];
105
- }
106
-
107
- function summarizedRangeOf(v: unknown): SummarizedRange | undefined {
108
- if (!isRecord(v)) return undefined;
109
- const summaryRef = stringOf(v['summaryRef']);
110
- const replacedTurns = stringArrayOf(v['replacedTurns']);
111
- if (summaryRef === undefined || replacedTurns === undefined) return undefined;
112
- return { summaryRef, replacedTurns };
113
- }
114
-
115
- /** Parse the seam response into a typed window — undefined if the shape is wrong. */
116
- function transcriptWindowOf(v: unknown): TranscriptWindow | undefined {
117
- if (!isRecord(v)) return undefined;
118
- const tokenCounter = stringOf(v['tokenCounter']);
119
- const tokenCount = numberOf(v['tokenCount']);
120
- const eventIds = stringArrayOf(v['eventIds']);
121
- if (tokenCounter === undefined || tokenCount === undefined || eventIds === undefined) return undefined;
122
- const rawRanges = v['summarizedRanges'];
123
- const summarizedRanges: SummarizedRange[] = [];
124
- if (Array.isArray(rawRanges)) {
125
- for (const r of rawRanges) {
126
- const parsed = summarizedRangeOf(r);
127
- if (parsed === undefined) return undefined; // malformed range → fail loudly via caller
128
- summarizedRanges.push(parsed);
129
- }
130
- }
131
- return { tokenCounter, tokenCount, eventIds, summarizedRanges };
84
+ const r = isRecord(v) ? v['runId'] : undefined;
85
+ return typeof r === 'string' ? r : undefined;
132
86
  }
133
87
 
134
88
  describe('context-budget-transcript-bound (RFC 0111 §"Context economy")', () => {
135
89
  it('bounds the per-turn transcript to transcriptTokenBudget with an internally-consistent, recent-tail accounting', async () => {
136
- const ma = await readCapabilityFamily<MultiAgentCap>('multiAgent');
137
- const cb = ma?.executionModel?.contextBudget;
138
- const budget = numberOf(cb?.transcriptTokenBudget);
139
- if (!behaviorGate(PROFILE, budget !== undefined)) return;
140
- if (!isFixtureAdvertised(FIXTURE)) return softSkip('inapplicable', 'capability or profile not advertised by this host — gate `!isFixtureAdvertised(FIXTURE)` returned early (fixture-gated soft-skip)'); // fixture-gated soft-skip
141
-
142
- const advertisedCounter = stringOf(cb?.tokenCounter);
143
- expect(
144
- advertisedCounter,
145
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111', 'tokenCounter MUST be advertised when transcriptTokenBudget is present (schema if/then)'),
146
- ).toBeDefined();
147
-
148
- // Drive the multi-turn orchestrator run.
149
- const create = await driver.post('/v1/runs', { workflowId: FIXTURE });
150
- expect(create.status).toBe(201);
90
+ const major = targetMajor();
91
+ let cb: Record<string, unknown> | undefined;
92
+ if (major === 2) {
93
+ const doc = await v2Discovery();
94
+ if (!doc) return softSkip('blocked', 'v2 discovery unreachable');
95
+ cb = recordAt(await familyAdvertised('multiAgent'), 'executionModel', 'contextBudget');
96
+ if (typeof cb?.['transcriptTokenBudget'] !== 'number') return softSkip('inapplicable', 'multiAgent.executionModel.contextBudget.transcriptTokenBudget is not advertised at major 2');
97
+ if (!seamsProfileAdvertised(doc)) return softSkip('inapplicable', 'the transcript-window seam is in the seams profile — conformance.seamsProfile is not openwop-conformance-seams-v2');
98
+ } else {
99
+ cb = recordAt(await readCapabilityFamily<Record<string, unknown>>('multiAgent'), 'executionModel', 'contextBudget');
100
+ if (!behaviorGate(PROFILE, typeof cb?.['transcriptTokenBudget'] === 'number')) return;
101
+ }
102
+ const budget = cb?.['transcriptTokenBudget'];
103
+ const advertisedCounter = cb?.['tokenCounter'];
104
+ if (typeof budget !== 'number') return softSkip('inapplicable', 'transcriptTokenBudget not advertised');
105
+ if (!isFixtureAdvertised(FIXTURE)) return softSkip('inapplicable', `the live fixture ${FIXTURE} is not advertised — the scripted multiturn fixture drives a mock supervisor, which RFC 0111 §Scope forbids from advertising contextBudget`);
106
+ expect(typeof advertisedCounter === 'string', req(ID, 'RFC 0111', 'tokenCounter MUST be advertised when transcriptTokenBudget is present (schema if/then)')).toBe(true);
107
+ if (typeof advertisedCounter !== 'string') return softSkip('blocked', 'contextBudget.tokenCounter is not advertised (the assertion above records the failure)');
108
+
109
+ const create = await driver.post(runsPath(), { workflowId: FIXTURE });
110
+ expect(create.status, req(ID, 'RFC 0111', `POST ${runsPath()} MUST create the live-fixture run`)).toBe(201);
151
111
  const runId = runIdOf(create.json);
152
- expect(runId, req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111', 'POST /v1/runs MUST return a runId')).toBeDefined();
153
- if (runId === undefined) return softSkip('blocked', 'precondition not met — `runId === undefined` returned early (seam, prior step, or fixture unavailable)');
112
+ expect(runId, req(ID, 'RFC 0111', 'the create response MUST carry a runId')).toBeDefined();
113
+ if (runId === undefined) return softSkip('blocked', 'no runId');
154
114
  await pollUntilTerminal(runId);
155
115
 
156
- // Probe the per-iteration transcript-window seam (OPTIONAL).
157
116
  const windows: Array<{ iteration: number; window: TranscriptWindow }> = [];
158
117
  for (let iteration = 1; iteration <= MAX_ITERATIONS_PROBED; iteration += 1) {
159
- const res = await driver.get(
160
- `/v1/host/sample/agent/transcript-window?runId=${encodeURIComponent(runId)}&iteration=${iteration}`,
161
- );
118
+ const res = await driver.get(`/v1/host/sample/agent/transcript-window?runId=${encodeURIComponent(runId)}&iteration=${iteration}`);
162
119
  if (res.status === 404 || res.status === 405) {
163
- if (iteration === 1) return softSkip('blocked', 'precondition not met — `iteration === 1` returned early (seam unwired — soft-skip the whole scenario) (seam, prior step, or fixture unavailable)'); // seam unwired — soft-skip the whole scenario
164
- break; // iterations exhausted
120
+ if (iteration === 1) return major === 2 ? seamAbsent(`contextBudget is advertised but the transcript-window seam answered ${res.status} (host-sample-test-seams.md §14)`) : softSkip('blocked', `the transcript-window seam answered ${res.status} (host-sample-test-seams.md §14)`);
121
+ break;
165
122
  }
166
- if (res.status === 400 || res.status === 422) break; // iteration past the run's last turn
167
- expect(
168
- res.status === 200,
169
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'host-sample-test-seams.md §14', 'the transcript-window seam MUST return 200 for a valid iteration'),
170
- ).toBe(true);
171
- const window = transcriptWindowOf(res.json);
172
- expect(
173
- window,
174
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'host-sample-test-seams.md §14', 'the seam MUST return { tokenCounter, tokenCount, eventIds, summarizedRanges }'),
175
- ).toBeDefined();
176
- if (window === undefined) return softSkip('blocked', 'precondition not met — `window === undefined` returned early (seam, prior step, or fixture unavailable)');
123
+ if (res.status === 400 || res.status === 422) break;
124
+ expect(res.status, req(ID, 'host-sample-test-seams.md §14', `iteration ${iteration}: the transcript-window seam MUST return 200 for a valid iteration`)).toBe(200);
125
+ const window = parseTranscriptWindow(res.json);
126
+ expect(window, req(ID, 'host-sample-test-seams.md §14', `iteration ${iteration}: the seam MUST return { tokenCounter, tokenCount, eventIds, summarizedRanges, entries? } with a well-formed entries[] when present`)).toBeDefined();
127
+ if (window === undefined) return softSkip('blocked', `iteration ${iteration}: the seam answer was malformed (the assertion above records the failure)`);
177
128
  windows.push({ iteration, window });
178
129
  }
130
+ expect(windows.length, req(ID, 'host-sample-test-seams.md §14', 'a wired transcript-window seam MUST report at least one orchestrator iteration')).toBeGreaterThan(0);
179
131
 
180
- // Non-vacuity: a wired seam MUST report at least one iteration.
181
- expect(windows.length, req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'host-sample-test-seams.md §14', 'a wired transcript-window seam MUST report at least one orchestrator iteration')).toBeGreaterThan(0);
182
-
183
- // Independent event-log read for the cross-check (OPTIONAL seam).
184
132
  const q = await queryTestEvents(runId);
185
- const logEventIds = new Set<string>();
186
- const summarizedRefs = new Set<string>();
187
- if (q.ok) {
188
- for (const e of q.events) {
189
- logEventIds.add(e.eventId);
190
- if (e.type === 'context.summarized') {
191
- const ref = stringOf(e.payload['summaryRef']);
192
- if (ref !== undefined) summarizedRefs.add(ref);
193
- }
194
- }
195
- }
196
-
133
+ const log: LogEvent[] | null = q.ok
134
+ ? q.events.map((e) => ({ eventId: e.eventId, type: e.type, sequence: e.sequence, payload: e.payload, ...(e.nodeId !== undefined ? { nodeId: e.nodeId } : {}) }))
135
+ : null;
136
+ const cap = { transcriptTokenBudget: budget, tokenCounter: advertisedCounter };
137
+ let pressure = false;
197
138
  for (const { iteration, window } of windows) {
198
- // 1 — tokenCounter agreement.
199
- expect(
200
- window.tokenCounter,
201
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111', `iteration ${iteration}: seam tokenCounter MUST equal the advertised contextBudget.tokenCounter`),
202
- ).toBe(advertisedCounter);
203
-
204
- // 2 — the per-turn token bound.
205
- if (budget !== undefined) {
206
- expect(
207
- window.tokenCount,
208
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111', `iteration ${iteration}: tokenCount MUST NOT exceed transcriptTokenBudget`),
209
- ).toBeLessThanOrEqual(budget);
210
- }
211
-
212
- // 3 — internal consistency: every named id is a real persisted event.
213
- if (q.ok) {
214
- for (const id of window.eventIds) {
215
- expect(
216
- logEventIds.has(id),
217
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111 §"Conformance seam"', `iteration ${iteration}: eventId "${id}" in the seam accounting MUST be a real persisted run event`),
218
- ).toBe(true);
219
- }
220
- }
221
-
222
- // 4 — recent-tail: ids are unique (no double-count inflating the window).
223
- const uniqueIds = new Set(window.eventIds);
224
- expect(
225
- uniqueIds.size,
226
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111 §"Conformance seam"', `iteration ${iteration}: eventIds MUST be a tail with no repeated entry`),
227
- ).toBe(window.eventIds.length);
228
-
229
- // 5 — every summarized range references a recorded context.summarized event.
230
- if (q.ok) {
231
- for (const range of window.summarizedRanges) {
232
- expect(
233
- summarizedRefs.has(range.summaryRef),
234
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111', `iteration ${iteration}: summarizedRanges summaryRef "${range.summaryRef}" MUST have a matching context.summarized event`),
235
- ).toBe(true);
236
- }
237
- }
139
+ const v = checkTranscriptWindow(iteration, window, cap, log);
140
+ expect(v.violations, req(ID, 'RFC 0111 §"Conformance seam" (b)/(c), host-sample-test-seams.md §14', `iteration ${iteration}: the host's transcript accounting MUST be within budget, recountable, real, and a recent tail`)).toEqual([]);
141
+ pressure ||= v.pressure;
238
142
  }
239
143
 
240
144
  // keepLastTurns verbatim — a kept turn is fed verbatim, never inside a summarized range.
241
- const keepLastTurns = numberOf(cb?.summarization?.keepLastTurns);
242
- if (keepLastTurns !== undefined && keepLastTurns > 0 && windows.length > 0) {
145
+ const keepLastTurns = recordAt(cb, 'summarization')?.['keepLastTurns'];
146
+ if (typeof keepLastTurns === 'number' && keepLastTurns > 0) {
243
147
  const last = windows[windows.length - 1].window;
244
- const summarizedIds = new Set<string>();
245
- for (const range of last.summarizedRanges) for (const id of range.replacedTurns) summarizedIds.add(id);
246
- const verbatimTail = last.eventIds.slice(Math.max(0, last.eventIds.length - keepLastTurns));
247
- for (const id of verbatimTail) {
248
- expect(
249
- summarizedIds.has(id),
250
- req('openwop.it.context-budget-transcript-bound.bounds-the-per-turn-transcript-to-transcripttokenbudget-with-an-internally-consi', 'RFC 0111', `a kept (verbatim) turn "${id}" MUST NOT appear inside a summarized range`),
251
- ).toBe(false);
148
+ const summarizedIds = new Set(last.summarizedRanges.flatMap((r) => r.replacedTurns));
149
+ for (const id of last.eventIds.slice(Math.max(0, last.eventIds.length - keepLastTurns))) {
150
+ expect(summarizedIds.has(id), req(ID, 'RFC 0111', `a kept (verbatim) turn "${id}" MUST NOT appear inside a summarized range`)).toBe(false);
252
151
  }
253
152
  }
153
+
154
+ if (log === null) return softSkip('blocked', 'the run event-log seam is unavailable, so the real-event, recent-tail and pressure rules were not measured');
155
+ if (!pressure) softSkip('inapplicable', `no iteration shows budget pressure — every eligible event fit under transcriptTokenBudget ${budget}, so the bound was never exercised (a budget the run never reaches is not a witness)`);
254
156
  });
255
157
  });