@openwop/openwop-conformance 2.42.3 → 2.42.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/README.md +1 -1
- package/dist/spec-artifacts.lock.json +2 -2
- package/package.json +2 -2
- package/requirements.json +292 -103
- package/schemas/CORPUS-STAMP.json +9 -9
- package/src/lib/agentOrgChart.ts +21 -4
- package/src/lib/effect-receiver.ts +23 -2
- package/src/lib/toolCatalog.ts +29 -5
- package/src/scenarios/agent-manifest-runtime.test.ts +17 -4
- package/src/scenarios/agent-org-chart-scoping.test.ts +27 -4
- package/src/scenarios/agent-roster-attribution.test.ts +28 -4
- package/src/scenarios/agentPackExport.test.ts +33 -0
- package/src/scenarios/agentPackInstall.test.ts +32 -0
- package/src/scenarios/agentReasoningEvents.test.ts +22 -13
- package/src/scenarios/agents-run-tool-allowlist.test.ts +6 -2
- package/src/scenarios/ai-envelope-shape.test.ts +9 -8
- package/src/scenarios/anonymous-actor-write-gated.test.ts +12 -6
- package/src/scenarios/audit-log-integrity.test.ts +37 -0
- package/src/scenarios/context-budget-transcript-bound.test.ts +14 -4
- package/src/scenarios/context-summarization-replay.test.ts +17 -8
- package/src/scenarios/conversationLifecycle.test.ts +12 -1
- package/src/scenarios/conversationReplayDeterminism.test.ts +19 -3
- package/src/scenarios/cost-attribution.test.ts +12 -0
- package/src/scenarios/debug-bundle-truncation.test.ts +16 -3
- package/src/scenarios/distillation-shape.test.ts +30 -0
- package/src/scenarios/envelope-truncated.test.ts +6 -2
- package/src/scenarios/feedback-fork-not-copied.test.ts +28 -7
- package/src/scenarios/idempotency-concurrent-claim.test.ts +27 -9
- package/src/scenarios/idempotency-key-determinism.test.ts +6 -2
- package/src/scenarios/mcp-server-untrusted-args.test.ts +20 -5
- package/src/scenarios/mcp-tool-roundtrip.test.ts +7 -21
- package/src/scenarios/memory-compaction-provenance-tag.test.ts +11 -3
- package/src/scenarios/orchestratorDispatch.test.ts +18 -2
- package/src/scenarios/pack-registry.test.ts +8 -1
- package/src/scenarios/production-backpressure.test.ts +17 -11
- package/src/scenarios/redaction.test.ts +8 -2
- package/src/scenarios/replay-fanout-suppression.test.ts +18 -2
- package/src/scenarios/runner-ledger.test.ts +7 -2
- package/src/scenarios/stream-modes.test.ts +27 -0
- package/src/scenarios/tool-catalog-projection.test.ts +35 -12
- package/src/scenarios/v2-effect-identity-business-key.test.ts +40 -6
- package/src/scenarios/v2-era-2-append-vocabulary.test.ts +12 -3
- package/src/scenarios/v2-idempotency-in-flight.test.ts +16 -0
- package/src/scenarios/v2-mcp-tasks.test.ts +17 -3
- package/src/scenarios/v2-min-client-version.test.ts +11 -1
- package/src/scenarios/v2-run-diff-identical.test.ts +13 -5
- package/src/scenarios/v2-run-pause-resume.test.ts +4 -1
- package/src/scenarios/v2-sse-last-event-id.test.ts +7 -0
- package/src/scenarios/v2-version-header-honored.test.ts +22 -3
- package/src/scenarios/v2-webhook-durable-delivery.test.ts +8 -0
- package/src/scenarios/v2-webhook-unregister-stops-delivery.test.ts +6 -1
- package/src/scenarios/version-negotiation.test.ts +12 -1
- package/src/scenarios/webhook-signed-delivery.test.ts +18 -3
|
@@ -21,6 +21,30 @@ import { behaviorGate } from '../lib/behavior-gate.js';
|
|
|
21
21
|
import { capabilityFamily } from '../lib/discovery-capabilities.js';
|
|
22
22
|
import { req } from '../lib/requirement-ids.js';
|
|
23
23
|
import { softSkip } from '../lib/soft-skip.js';
|
|
24
|
+
import Ajv2020 from 'ajv/dist/2020.js';
|
|
25
|
+
import { readFileSync } from 'node:fs';
|
|
26
|
+
import { join } from 'node:path';
|
|
27
|
+
import { SCHEMAS_DIR } from '../lib/paths.js';
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* audit-verify-result.schema.json with every `additionalProperties: false`
|
|
31
|
+
* relaxed: the REQUIRED members and their types are enforced, extra members
|
|
32
|
+
* are not (the reference postgres host emits `checkpointsValid` and a
|
|
33
|
+
* per-checkpoint `verified` bit; convicting extras is a separate decision).
|
|
34
|
+
*/
|
|
35
|
+
function relaxAdditional(node: unknown): unknown {
|
|
36
|
+
if (Array.isArray(node)) return node.map(relaxAdditional);
|
|
37
|
+
if (node === null || typeof node !== 'object') return node;
|
|
38
|
+
const out: Record<string, unknown> = {};
|
|
39
|
+
for (const [k, v] of Object.entries(node as Record<string, unknown>)) {
|
|
40
|
+
if (k === 'additionalProperties' && v === false) continue;
|
|
41
|
+
out[k] = relaxAdditional(v);
|
|
42
|
+
}
|
|
43
|
+
return out;
|
|
44
|
+
}
|
|
45
|
+
const VERIFY_RESULT_SCHEMA = relaxAdditional(
|
|
46
|
+
JSON.parse(readFileSync(join(SCHEMAS_DIR, 'audit-verify-result.schema.json'), 'utf8')),
|
|
47
|
+
) as Record<string, unknown>;
|
|
24
48
|
|
|
25
49
|
interface AuditIntegrityCaps {
|
|
26
50
|
hashChain?: boolean;
|
|
@@ -90,6 +114,19 @@ describe('audit-log-integrity: verify endpoint returns chainValid', () => {
|
|
|
90
114
|
anomalies?: unknown[];
|
|
91
115
|
};
|
|
92
116
|
|
|
117
|
+
// unfailable-leg audit wave 2, 2026-09-27: a body that OMITTED
|
|
118
|
+
// `checkpoints` (or sent malformed checkpoint objects) passed — the
|
|
119
|
+
// signature check below ran only `if (Array.isArray(body.checkpoints) &&
|
|
120
|
+
// length > 0)`. The body is now validated against
|
|
121
|
+
// audit-verify-result.schema.json (checkpoints[] + anomalies[] required;
|
|
122
|
+
// each checkpoint needs checkpoint/atSequence/merkleRoot/signature).
|
|
123
|
+
const validate = new Ajv2020({ strict: false, allErrors: true }).compile(VERIFY_RESULT_SCHEMA);
|
|
124
|
+
const shapeOk = validate(body);
|
|
125
|
+
expect(shapeOk, req('openwop.it.audit-log-integrity.get-v1-audit-verify-on-a-recent-range-reports-chainvalid-true',
|
|
126
|
+
'audit-verify-result.schema.json',
|
|
127
|
+
`GET /v1/audit/verify MUST return the AuditVerifyResult shape: ${JSON.stringify(validate.errors ?? [])}`,
|
|
128
|
+
)).toBe(true);
|
|
129
|
+
|
|
93
130
|
expect(body.chainValid, req('openwop.it.audit-log-integrity.get-v1-audit-verify-on-a-recent-range-reports-chainvalid-true',
|
|
94
131
|
'auth-profiles.md §"Audit-log integrity"',
|
|
95
132
|
'unmodified audit range MUST report chainValid: true',
|
|
@@ -61,7 +61,7 @@ import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
|
61
61
|
import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
|
|
62
62
|
import { queryTestEvents } from '../lib/event-log-query.js';
|
|
63
63
|
import { req } from '../lib/requirement-ids.js';
|
|
64
|
-
import { softSkip, seamAbsent } from '../lib/soft-skip.js';
|
|
64
|
+
import { blockedDespiteAssertions, softSkip, seamAbsent } from '../lib/soft-skip.js';
|
|
65
65
|
import { seamsProfileAdvertised, targetMajor } from '../lib/seams.js';
|
|
66
66
|
import { familyAdvertised, v2Discovery } from '../lib/v2.js';
|
|
67
67
|
import { runsPath } from '../lib/memoryAttribution.js';
|
|
@@ -117,7 +117,11 @@ describe('context-budget-transcript-bound (RFC 0111 §"Context economy")', () =>
|
|
|
117
117
|
for (let iteration = 1; iteration <= MAX_ITERATIONS_PROBED; iteration += 1) {
|
|
118
118
|
const res = await driver.get(`/v1/host/sample/agent/transcript-window?runId=${encodeURIComponent(runId)}&iteration=${iteration}`);
|
|
119
119
|
if (res.status === 404 || res.status === 405) {
|
|
120
|
-
|
|
120
|
+
// unfailable-leg audit wave 2, 2026-09-27: the major-1 branch was
|
|
121
|
+
// `softSkip('blocked')` after the create/runId asserts, which records a
|
|
122
|
+
// partial-witness PASS — a host with no transcript-window seam passed
|
|
123
|
+
// the transcript bound without a single window being read.
|
|
124
|
+
if (iteration === 1) return major === 2 ? seamAbsent(`contextBudget is advertised but the transcript-window seam answered ${res.status} (host-sample-test-seams.md §14)`) : blockedDespiteAssertions(`the transcript-window seam answered ${res.status} (host-sample-test-seams.md §14)`);
|
|
121
125
|
break;
|
|
122
126
|
}
|
|
123
127
|
if (res.status === 400 || res.status === 422) break;
|
|
@@ -151,7 +155,13 @@ describe('context-budget-transcript-bound (RFC 0111 §"Context economy")', () =>
|
|
|
151
155
|
}
|
|
152
156
|
}
|
|
153
157
|
|
|
154
|
-
|
|
155
|
-
|
|
158
|
+
// unfailable-leg audit wave 2, 2026-09-27: both returns below were plain
|
|
159
|
+
// soft-skips after assertions, i.e. partial-witness PASSES. A host whose
|
|
160
|
+
// event log was unreadable (real-event / recent-tail rules unmeasured), or
|
|
161
|
+
// whose run never put the budget under pressure (the bound never
|
|
162
|
+
// exercised — any host "fits" a budget it never reaches), certified the
|
|
163
|
+
// transcript bound it was never observed to enforce.
|
|
164
|
+
if (log === null) return blockedDespiteAssertions('the run event-log seam is unavailable, so the real-event, recent-tail and pressure rules were not measured');
|
|
165
|
+
if (!pressure) return blockedDespiteAssertions(`no iteration shows budget pressure — every eligible event fit under transcriptTokenBudget ${budget}, so the bound was never exercised (a budget the run never reaches is not a witness)`);
|
|
156
166
|
}, liveScenarioTimeoutMs(1));
|
|
157
167
|
});
|
|
@@ -44,7 +44,7 @@ import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
|
44
44
|
import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
|
|
45
45
|
import { queryTestEvents, type TestEvent } from '../lib/event-log-query.js';
|
|
46
46
|
import { req } from '../lib/requirement-ids.js';
|
|
47
|
-
import { softSkip } from '../lib/soft-skip.js';
|
|
47
|
+
import { softSkip, blockedDespiteAssertions } from '../lib/soft-skip.js';
|
|
48
48
|
import { seamsProfileAdvertised, targetMajor } from '../lib/seams.js';
|
|
49
49
|
import { familyAdvertised, v2Discovery } from '../lib/v2.js';
|
|
50
50
|
import { runsPath } from '../lib/memoryAttribution.js';
|
|
@@ -120,31 +120,40 @@ describe('context-summarization-replay (RFC 0111 §"Replay determinism")', () =>
|
|
|
120
120
|
}
|
|
121
121
|
if (!isFixtureAdvertised(FIXTURE)) return softSkip('inapplicable', `the live fixture ${FIXTURE} is not advertised — RFC 0111 §Scope forbids a mock supervisor from advertising contextBudget`);
|
|
122
122
|
|
|
123
|
+
// unfailable-leg audit wave 2, 2026-09-27: the replay-mode gate moved ABOVE the
|
|
124
|
+
// create — after the 201 assert it recorded a partial-witness `executed-pass`
|
|
125
|
+
// for a host that never replayed anything.
|
|
126
|
+
if (!replayModesOf(wellKnown).includes('replay')) return softSkip('inapplicable', 'the host advertises no replay fork mode');
|
|
127
|
+
|
|
123
128
|
const create = await driver.post(runsPath(), { workflowId: FIXTURE });
|
|
124
129
|
expect(create.status, req(ID, 'RFC 0111', `POST ${runsPath()} MUST create the live-fixture run`)).toBe(201);
|
|
125
130
|
const sourceRunId = runIdOf(create.json);
|
|
126
131
|
expect(sourceRunId, req(ID, 'rest-endpoints.md POST /v1/runs', 'the create response MUST carry a runId')).toBeDefined();
|
|
127
|
-
if (sourceRunId === undefined) return
|
|
132
|
+
if (sourceRunId === undefined) return blockedDespiteAssertions('no runId');
|
|
128
133
|
await pollUntilTerminal(sourceRunId, { timeoutMs: LIVE_RUN_POLL_MS });
|
|
134
|
+
// unfailable-leg audit wave 2, 2026-09-27: every early return below follows the
|
|
135
|
+
// create's 201 assert, so a plain softSkip recorded a partial-witness
|
|
136
|
+
// `executed-pass` — a host whose event log was unreadable, that never
|
|
137
|
+
// summarized, or whose fork 404'd passed without the replay requirement
|
|
138
|
+
// (reuse of the recorded summaryRef) ever being observed. Now `blocked`.
|
|
129
139
|
|
|
130
140
|
const sourceQ = await queryTestEvents(sourceRunId);
|
|
131
|
-
if (!sourceQ.ok) return
|
|
141
|
+
if (!sourceQ.ok) return blockedDespiteAssertions('the run event-log seam is unavailable — summaryRef reuse unobserved');
|
|
132
142
|
const sourceFingerprints = summaryFingerprints(sourceQ.events);
|
|
133
143
|
if (sourceFingerprints.length === 0) {
|
|
134
|
-
return
|
|
144
|
+
return blockedDespiteAssertions('the live run produced no context.summarized event — a host advertising summarization must summarize this fixture for reuse to be observable');
|
|
135
145
|
}
|
|
136
|
-
if (!replayModesOf(wellKnown).includes('replay')) return softSkip('inapplicable', 'the host advertises no replay fork mode');
|
|
137
146
|
|
|
138
147
|
const fork = await driver.post(`${runsPath()}/${encodeURIComponent(sourceRunId)}:fork`, { fromSeq: 0, mode: 'replay' });
|
|
139
|
-
if (fork.status === 501 || fork.status === 404) return
|
|
148
|
+
if (fork.status === 501 || fork.status === 404) return blockedDespiteAssertions(`replay fork answered ${fork.status} — summaryRef reuse unobserved`);
|
|
140
149
|
expect(fork.status, req(ID, 'rest-endpoints.md POST /v1/runs/{runId}:fork', 'replay fork MUST return 201')).toBe(201);
|
|
141
150
|
const forkRunId = runIdOf(fork.json);
|
|
142
151
|
expect(forkRunId, req(ID, 'rest-endpoints.md POST /v1/runs/{runId}:fork', 'replay fork MUST return a runId')).toBeDefined();
|
|
143
|
-
if (forkRunId === undefined) return
|
|
152
|
+
if (forkRunId === undefined) return blockedDespiteAssertions('no fork runId');
|
|
144
153
|
await pollUntilTerminal(forkRunId, { timeoutMs: LIVE_RUN_POLL_MS });
|
|
145
154
|
|
|
146
155
|
const forkQ = await queryTestEvents(forkRunId);
|
|
147
|
-
if (!forkQ.ok) return
|
|
156
|
+
if (!forkQ.ok) return blockedDespiteAssertions('the event-log seam is unavailable for the fork — summaryRef reuse unobserved');
|
|
148
157
|
expect(summaryFingerprints(forkQ.events), req(ID, 'RFC 0111 §"Replay determinism"', 'a replay fork MUST reuse the recorded context.summarized summaryRef (never re-summarize to a different transcript)')).toEqual(sourceFingerprints);
|
|
149
158
|
|
|
150
159
|
// The model-facing half: the summary TEXT the host fed on the fork.
|
|
@@ -35,7 +35,13 @@ describe.skipIf(SKIP)('conversationLifecycle: open → exchange → close round-
|
|
|
35
35
|
|
|
36
36
|
// The fixture's exchange step requires resume input. Host-internal
|
|
37
37
|
// mock auto-resumes for conformance.
|
|
38
|
-
await pollUntilTerminal(runId);
|
|
38
|
+
const terminal = await pollUntilTerminal(runId);
|
|
39
|
+
// unfailable-leg audit wave 2, 2026-09-27: the leg never checked the
|
|
40
|
+
// exchange happened or the run finished — a host emitting only
|
|
41
|
+
// `conversation.opened` + `conversation.closed` (zero exchanges, so the CO-3
|
|
42
|
+
// "no exchange after close" check is vacuous) in a run that FAILED passed.
|
|
43
|
+
// The fixture performs exactly one exchange (mockAutoResume) and completes.
|
|
44
|
+
expect(terminal.status, req('openwop.it.conversationLifecycle.emits-all-three-lifecycle-events-with-matching-conversationid-no-exchanges-after', 'RFCS/0005-conversation.md', 'the open → exchange → close fixture run MUST terminate `completed`')).toBe('completed');
|
|
39
45
|
|
|
40
46
|
const events = await driver.get(`/v1/runs/${encodeURIComponent(runId)}/events`);
|
|
41
47
|
const list = (events.json as { events?: Array<{ type: string; payload?: Record<string, unknown> }> })
|
|
@@ -47,6 +53,7 @@ describe.skipIf(SKIP)('conversationLifecycle: open → exchange → close round-
|
|
|
47
53
|
|
|
48
54
|
expect(opened.length).toBeGreaterThan(0);
|
|
49
55
|
expect(closed.length).toBeGreaterThan(0);
|
|
56
|
+
expect(exchanged.length, req('openwop.it.conversationLifecycle.emits-all-three-lifecycle-events-with-matching-conversationid-no-exchanges-after', 'RFCS/0005-conversation.md', '`conversation.exchanged` MUST be emitted for the fixture\'s one exchange turn')).toBeGreaterThan(0);
|
|
50
57
|
|
|
51
58
|
// All three event types MUST share the same conversationId for the
|
|
52
59
|
// fixture's single conversation.
|
|
@@ -62,5 +69,9 @@ describe.skipIf(SKIP)('conversationLifecycle: open → exchange → close round-
|
|
|
62
69
|
.slice(closedIdx + 1)
|
|
63
70
|
.filter((e) => e.type === 'conversation.exchanged' && e.payload?.conversationId === convId);
|
|
64
71
|
expect(exchangedAfterClose.length).toBe(0);
|
|
72
|
+
// ...and the exchange this run DID emit precedes the close (open → exchange → close).
|
|
73
|
+
let lastExchangedIdx = -1;
|
|
74
|
+
list.forEach((e, i) => { if (e.type === 'conversation.exchanged' && e.payload?.conversationId === convId) lastExchangedIdx = i; });
|
|
75
|
+
expect(lastExchangedIdx >= 0 && lastExchangedIdx < closedIdx, req('openwop.it.conversationLifecycle.emits-all-three-lifecycle-events-with-matching-conversationid-no-exchanges-after', 'RFCS/0005-conversation.md (CO-3)', `the last conversation.exchanged (index ${lastExchangedIdx}) MUST precede conversation.closed (index ${closedIdx})`)).toBe(true);
|
|
65
76
|
});
|
|
66
77
|
});
|
|
@@ -21,12 +21,22 @@ import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
|
21
21
|
import { isConversationPrimitiveSupported } from '../lib/multi-agent-capabilities.js';
|
|
22
22
|
import { softSkip } from '../lib/soft-skip.js';
|
|
23
23
|
import { req } from '../lib/requirement-ids.js';
|
|
24
|
+
import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
|
|
24
25
|
|
|
25
26
|
const FIXTURE = 'conformance-conversation-replay';
|
|
26
27
|
const SKIP = !isConversationPrimitiveSupported() || !isFixtureAdvertised(FIXTURE);
|
|
27
28
|
|
|
28
29
|
describe.skipIf(SKIP)('conversationReplayDeterminism: replay-fork preserves conversation log', () => {
|
|
29
30
|
it('forked run yields byte-equal conversation channel projection', async () => {
|
|
31
|
+
// unfailable-leg audit wave 2, 2026-09-27: the replay-fork gate used to be
|
|
32
|
+
// the fork response itself (404/501 → soft-skip 'inapplicable'), reached
|
|
33
|
+
// only AFTER the create/terminal asserts — a host with no fork surface
|
|
34
|
+
// recorded a partial-witness pass. Gate on the advertised `replay.modes`
|
|
35
|
+
// BEFORE any assertion; once `replay` is advertised, a 404/501 fork fails.
|
|
36
|
+
const replayCap = await readCapabilityFamily<{ supported?: unknown; modes?: unknown }>('replay');
|
|
37
|
+
const modes = replayCap?.supported === true && Array.isArray(replayCap.modes) ? replayCap.modes : [];
|
|
38
|
+
if (!modes.includes('replay')) return softSkip('inapplicable', 'host does not advertise replay.modes including "replay"');
|
|
39
|
+
|
|
30
40
|
const create = await driver.post('/v1/runs', { workflowId: FIXTURE });
|
|
31
41
|
expect(create.status, req('openwop.it.conversationReplayDeterminism.forked-run-yields-byte-equal-conversation-channel-projection', 'RFCS/0005-conversation.md', 'forked run yields byte-equal conversation channel projection')).toBe(201);
|
|
32
42
|
const sourceRunId = (create.json as { runId: string }).runId;
|
|
@@ -36,12 +46,18 @@ describe.skipIf(SKIP)('conversationReplayDeterminism: replay-fork preserves conv
|
|
|
36
46
|
|
|
37
47
|
const sourceSnap = await driver.get(`/v1/runs/${encodeURIComponent(sourceRunId)}`);
|
|
38
48
|
const sourceConv = (sourceSnap.json as { channels?: Record<string, unknown> }).channels;
|
|
49
|
+
// unfailable-leg audit wave 2, 2026-09-27: a host that projected NO
|
|
50
|
+
// channels on either run compared `undefined` to `undefined` and passed
|
|
51
|
+
// "byte-equal". The source projection must exist and be non-empty first.
|
|
52
|
+
expect(
|
|
53
|
+
sourceConv !== null && typeof sourceConv === 'object' && !Array.isArray(sourceConv) && Object.keys(sourceConv).length > 0,
|
|
54
|
+
req('openwop.it.conversationReplayDeterminism.forked-run-yields-byte-equal-conversation-channel-projection', 'RFCS/0005-conversation.md', 'the source run of a conversation fixture MUST project a non-empty channels object'),
|
|
55
|
+
).toBe(true);
|
|
39
56
|
|
|
40
57
|
const fork = await driver.post(`/v1/runs/${encodeURIComponent(sourceRunId)}:fork`, {
|
|
41
58
|
mode: 'replay',
|
|
42
59
|
});
|
|
43
|
-
|
|
44
|
-
expect([200, 201]).toContain(fork.status);
|
|
60
|
+
expect([200, 201], req('openwop.it.conversationReplayDeterminism.forked-run-yields-byte-equal-conversation-channel-projection', 'spec/v1/replay.md', 'a host advertising replay.modes ["replay"] MUST accept a mode:"replay" fork (404/501 is a failure)')).toContain(fork.status);
|
|
45
61
|
|
|
46
62
|
const forkedRunId = (fork.json as { runId: string }).runId;
|
|
47
63
|
const forkedTerminal = await pollUntilTerminal(forkedRunId);
|
|
@@ -50,6 +66,6 @@ describe.skipIf(SKIP)('conversationReplayDeterminism: replay-fork preserves conv
|
|
|
50
66
|
const forkedSnap = await driver.get(`/v1/runs/${encodeURIComponent(forkedRunId)}`);
|
|
51
67
|
const forkedConv = (forkedSnap.json as { channels?: Record<string, unknown> }).channels;
|
|
52
68
|
|
|
53
|
-
expect(JSON.stringify(forkedConv)).toBe(JSON.stringify(sourceConv));
|
|
69
|
+
expect(JSON.stringify(forkedConv), req('openwop.it.conversationReplayDeterminism.forked-run-yields-byte-equal-conversation-channel-projection', 'RFCS/0005-conversation.md', 'a replay-forked run MUST yield a byte-equal conversation channel projection')).toBe(JSON.stringify(sourceConv));
|
|
54
70
|
});
|
|
55
71
|
});
|
|
@@ -163,6 +163,18 @@ describe.skipIf(SKIP_NO_COST_EMIT)('cost-attribution: end-to-end roundtrip via c
|
|
|
163
163
|
'metrics.openwopCost MUST be populated after a node calls ctx.recordCost()',
|
|
164
164
|
)).toBeDefined();
|
|
165
165
|
|
|
166
|
+
// unfailable-leg audit wave 2, 2026-09-27: every field check below is
|
|
167
|
+
// `if (field in openwopCost)` — a host returning `openwopCost: {}` (or one
|
|
168
|
+
// carrying only provider/model) passed the whole leg with no cost recorded.
|
|
169
|
+
// The canary records a cost, so at least one of usd / tokens MUST surface.
|
|
170
|
+
expect(
|
|
171
|
+
(openwopCost !== undefined && openwopCost !== null) && ('usd' in openwopCost || ('tokens' in openwopCost && openwopCost.tokens !== undefined && openwopCost.tokens !== null)),
|
|
172
|
+
req('openwop.it.cost-attribution.metrics-openwopcost-must-carry-the-canary-cost-shape-after-the-fixture-node-runs',
|
|
173
|
+
'run-snapshot.schema.json §metrics.openwopCost / observability.md §Cost attribution attributes',
|
|
174
|
+
'metrics.openwopCost MUST carry the recorded cost — at least one of usd or tokens — after ctx.recordCost()',
|
|
175
|
+
),
|
|
176
|
+
).toBe(true);
|
|
177
|
+
|
|
166
178
|
// Provider — the fixture canary is a stable string. Host-defined
|
|
167
179
|
// overrides are spec-allowed; we assert shape rather than exact match.
|
|
168
180
|
if ('provider' in openwopCost!) {
|
|
@@ -24,7 +24,7 @@ import { driver } from '../lib/driver.js';
|
|
|
24
24
|
import { pollUntilTerminal } from '../lib/polling.js';
|
|
25
25
|
import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
26
26
|
import { req } from '../lib/requirement-ids.js';
|
|
27
|
-
import { softSkip } from '../lib/soft-skip.js';
|
|
27
|
+
import { softSkip, blockedDespiteAssertions } from '../lib/soft-skip.js';
|
|
28
28
|
|
|
29
29
|
// `conformance-multi-node` produces enough events (run.started, three
|
|
30
30
|
// node.started/completed pairs, run.completed = ~8 events) that
|
|
@@ -70,12 +70,25 @@ describe('debug-bundle-truncation: truncated: true contract', () => {
|
|
|
70
70
|
metrics?: { eventCount?: number };
|
|
71
71
|
};
|
|
72
72
|
|
|
73
|
-
|
|
73
|
+
// unfailable-leg audit wave 2, 2026-09-27: `truncated !== true` used to
|
|
74
|
+
// soft-skip unconditionally after the create/baseline asserts — a
|
|
75
|
+
// partial-witness PASS even for a host that DID cut the events array
|
|
76
|
+
// (fewer than the full bundle) but omitted `truncated: true`, which is
|
|
77
|
+
// exactly the violation this scenario exists to catch. Now: a shortened
|
|
78
|
+
// events array MUST carry `truncated: true`; an un-shortened bundle means
|
|
79
|
+
// the cap was never lowered, so the requirement was not observed.
|
|
80
|
+
const returnedCount = Array.isArray(body.events) ? body.events.length : fullEventCount;
|
|
81
|
+
if (returnedCount < fullEventCount) {
|
|
82
|
+
expect(body.truncated, req('openwop.it.debug-bundle-truncation.host-that-supports-maxevents-n-or-otherwise-caps-surfaces-truncated-truncatedrea',
|
|
83
|
+
'debug-bundle.md §"Bundle size limits"',
|
|
84
|
+
`a bundle whose events were cut (${returnedCount} of ${fullEventCount}) MUST set truncated: true`,
|
|
85
|
+
)).toBe(true);
|
|
86
|
+
} else if (body.truncated !== true) {
|
|
74
87
|
// eslint-disable-next-line no-console
|
|
75
88
|
console.warn(
|
|
76
89
|
'[debug-bundle-truncation] host does not honor ?maxEvents=; skipping truncated-shape assertions',
|
|
77
90
|
);
|
|
78
|
-
return
|
|
91
|
+
return blockedDespiteAssertions('host does not honor ?maxEvents= (full bundle returned, truncated not set) — truncation contract not observed');
|
|
79
92
|
}
|
|
80
93
|
|
|
81
94
|
expect(typeof body.truncatedReason, req('openwop.it.debug-bundle-truncation.host-that-supports-maxevents-n-or-otherwise-caps-surfaces-truncated-truncatedrea',
|
|
@@ -12,10 +12,30 @@
|
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
14
|
import { describe, it, expect } from 'vitest';
|
|
15
|
+
import { readFileSync } from 'node:fs';
|
|
16
|
+
import { join } from 'node:path';
|
|
17
|
+
import Ajv2020 from 'ajv/dist/2020.js';
|
|
18
|
+
import { SCHEMAS_DIR } from '../lib/paths.js';
|
|
15
19
|
import { softSkip } from '../lib/soft-skip.js';
|
|
16
20
|
import { readDistillationCap } from '../lib/distillation.js';
|
|
17
21
|
import { req } from '../lib/requirement-ids.js';
|
|
18
22
|
|
|
23
|
+
/**
|
|
24
|
+
* The `memory.distillation` subtree of `capabilities.schema.json` (major 1 —
|
|
25
|
+
* this file is `[1]` in scenario-majors.json). Self-contained (no `$ref`), so
|
|
26
|
+
* it compiles alone.
|
|
27
|
+
*/
|
|
28
|
+
function distillationBlockValidator(): (doc: unknown) => { ok: boolean; errors: string } {
|
|
29
|
+
const caps = JSON.parse(readFileSync(join(SCHEMAS_DIR, 'capabilities.schema.json'), 'utf8')) as {
|
|
30
|
+
properties?: { memory?: { properties?: { distillation?: Record<string, unknown> } } };
|
|
31
|
+
};
|
|
32
|
+
const sub = caps.properties?.memory?.properties?.distillation;
|
|
33
|
+
if (sub === undefined) throw new Error('capabilities.schema.json has no properties.memory.properties.distillation subtree');
|
|
34
|
+
const ajv = new Ajv2020({ allErrors: true, strict: false });
|
|
35
|
+
const validate = ajv.compile(sub);
|
|
36
|
+
return (doc: unknown) => ({ ok: validate(doc) as boolean, errors: ajv.errorsText(validate.errors, { separator: '; ' }) });
|
|
37
|
+
}
|
|
38
|
+
|
|
19
39
|
describe('distillation-shape: advertisement (RFC 0062 §A)', () => {
|
|
20
40
|
it('capabilities.memory.distillation is absent or a well-formed object', async () => {
|
|
21
41
|
const cap = await readDistillationCap();
|
|
@@ -30,6 +50,16 @@ describe('distillation-shape: advertisement (RFC 0062 §A)', () => {
|
|
|
30
50
|
req('openwop.it.distillation-shape.capabilities-memory-distillation-is-absent-or-a-well-formed-object', 'capabilities.schema.json §memory.distillation', 'maxTokenBudget MUST be a positive integer when present'),
|
|
31
51
|
).toBe(true);
|
|
32
52
|
}
|
|
53
|
+
// unfailable-leg audit wave 2, 2026-09-27: the hand checks above accept a
|
|
54
|
+
// fractional `maxTokenBudget` (1.5), a non-string `tokenizerName`, and any
|
|
55
|
+
// unknown key — the schema says `integer`, `string`, and
|
|
56
|
+
// `additionalProperties: false`. The whole block now validates against the
|
|
57
|
+
// capabilities.schema.json `memory.distillation` subtree.
|
|
58
|
+
const v = distillationBlockValidator()(cap);
|
|
59
|
+
expect(
|
|
60
|
+
v.ok,
|
|
61
|
+
req('openwop.it.distillation-shape.capabilities-memory-distillation-is-absent-or-a-well-formed-object', 'capabilities.schema.json §memory.distillation', `the distillation block MUST validate against the schema subtree (${v.errors})`),
|
|
62
|
+
).toBe(true);
|
|
33
63
|
for (const k of ['scheduled', 'indexEmitted'] as const) {
|
|
34
64
|
if (cap[k] !== undefined) {
|
|
35
65
|
expect(
|
|
@@ -21,7 +21,7 @@ import { driver } from '../lib/driver.js';
|
|
|
21
21
|
import { pollUntilTerminal } from '../lib/polling.js';
|
|
22
22
|
import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
23
23
|
import { req } from '../lib/requirement-ids.js';
|
|
24
|
-
import { softSkip } from '../lib/soft-skip.js';
|
|
24
|
+
import { softSkip, blockedDespiteAssertions } from '../lib/soft-skip.js';
|
|
25
25
|
|
|
26
26
|
const HTTP_SKIP = !process.env.OPENWOP_BASE_URL;
|
|
27
27
|
const FIXTURE = 'conformance-envelope-truncated';
|
|
@@ -67,7 +67,11 @@ describe.skipIf(HTTP_SKIP)('envelope-truncated: runtime behavior (RFC 0032 §B.4
|
|
|
67
67
|
expect(seed.status).toBe(200);
|
|
68
68
|
|
|
69
69
|
const result = await startRunAndRead();
|
|
70
|
-
|
|
70
|
+
// unfailable-leg audit wave 2, 2026-09-27: after the seed assertion above a
|
|
71
|
+
// plain softSkip('blocked') records a partial-witness PASS at major 1, so a
|
|
72
|
+
// host whose run create / event read failed passed this leg without the
|
|
73
|
+
// envelope.truncated count ever being observed.
|
|
74
|
+
if (result === null) return blockedDespiteAssertions('the run create or event-log read failed after the mock was seeded — the envelope.truncated count was never observed');
|
|
71
75
|
const truncated = result.events.filter((e) => e.type === 'envelope.truncated');
|
|
72
76
|
expect(
|
|
73
77
|
truncated.length,
|
|
@@ -8,7 +8,8 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import { describe, it, expect } from 'vitest';
|
|
11
|
-
import { softSkip } from '../lib/soft-skip.js';
|
|
11
|
+
import { softSkip, blockedDespiteAssertions } from '../lib/soft-skip.js';
|
|
12
|
+
import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
|
|
12
13
|
import { driver } from '../lib/driver.js';
|
|
13
14
|
import { pollUntilTerminal } from '../lib/polling.js';
|
|
14
15
|
import { readFeedbackCap, seedRun } from '../lib/feedback.js';
|
|
@@ -18,24 +19,44 @@ describe('feedback-fork-not-copied (RFC 0056 §D)', () => {
|
|
|
18
19
|
it('a fork of an annotated run starts with zero annotations', async () => {
|
|
19
20
|
const cap = await readFeedbackCap();
|
|
20
21
|
if (cap?.supported !== true) return softSkip('inapplicable', 'capability or profile not advertised by this host — gate `cap?.supported !== true` returned early');
|
|
22
|
+
// unfailable-leg audit wave 2, 2026-09-27: three holes, all passing a
|
|
23
|
+
// non-conforming host. (1) Fork support was discovered only AFTER the
|
|
24
|
+
// annotation 201 assert, so a host with no fork recorded a partial-witness
|
|
25
|
+
// pass — gate on advertised `replay.modes` ⊇ ['branch'] BEFORE asserting.
|
|
26
|
+
// (2) `annotations ?? []` read a 404/500/non-array list as "zero
|
|
27
|
+
// annotations" — the fork list must now be 200 + an array. (3) No positive
|
|
28
|
+
// control: a host whose annotation list NEVER returns anything passed —
|
|
29
|
+
// the SOURCE run must list the annotation just posted.
|
|
30
|
+
const replayCap = await readCapabilityFamily<{ supported?: unknown; modes?: unknown }>('replay');
|
|
31
|
+
const modes = replayCap?.supported === true && Array.isArray(replayCap.modes) ? replayCap.modes : [];
|
|
32
|
+
if (!modes.includes('branch')) return softSkip('inapplicable', 'host does not advertise replay.modes including "branch" — fork leg not applicable');
|
|
21
33
|
const runId = await seedRun('feedback-fork');
|
|
22
34
|
if (!runId) return softSkip('blocked', 'precondition not met — `!runId` returned early (seam, prior step, or fixture unavailable)');
|
|
23
35
|
const post = await driver.post(`/v1/runs/${runId}/annotations`, { signal: { kind: 'flag' } });
|
|
24
36
|
if (post.status === 501 || post.status === 404) return softSkip('blocked', 'precondition not met — `post.status === 501 || post.status === 404` returned early (seam, prior step, or fixture unavailable)');
|
|
25
|
-
expect(post.status).toBe(201);
|
|
37
|
+
expect(post.status, req('openwop.it.feedback-fork-not-copied.a-fork-of-an-annotated-run-starts-with-zero-annotations', 'RFC 0056 §B', 'POST /v1/runs/{runId}/annotations MUST return 201')).toBe(201);
|
|
26
38
|
try {
|
|
27
39
|
await pollUntilTerminal(runId, { timeoutMs: 10_000 });
|
|
28
40
|
} catch {
|
|
29
|
-
return
|
|
41
|
+
return blockedDespiteAssertions('source run did not reach a terminal state within 10s — fork not attempted');
|
|
30
42
|
}
|
|
43
|
+
const srcList = await driver.get(`/v1/runs/${runId}/annotations`);
|
|
44
|
+
expect(srcList.status, req('openwop.it.feedback-fork-not-copied.a-fork-of-an-annotated-run-starts-with-zero-annotations', 'RFC 0056 §B', 'GET /v1/runs/{runId}/annotations MUST return 200')).toBe(200);
|
|
45
|
+
const srcAnn = (srcList.json as { annotations?: unknown } | undefined)?.annotations;
|
|
46
|
+
expect(
|
|
47
|
+
Array.isArray(srcAnn) && srcAnn.length >= 1,
|
|
48
|
+
req('openwop.it.feedback-fork-not-copied.a-fork-of-an-annotated-run-starts-with-zero-annotations', 'RFC 0056 §B', 'positive control: the source run MUST list the annotation just posted'),
|
|
49
|
+
).toBe(true);
|
|
31
50
|
const fork = await driver.post(`/v1/runs/${runId}:fork`, { fromSeq: 0, mode: 'branch' });
|
|
32
|
-
if (fork.status !== 200 && fork.status !== 201) return
|
|
51
|
+
if (fork.status !== 200 && fork.status !== 201) return blockedDespiteAssertions(`branch fork declined (HTTP ${fork.status}) although replay.modes advertises "branch" — fork annotation state not observed`);
|
|
33
52
|
const forkId = (fork.json as { runId?: string } | undefined)?.runId;
|
|
34
|
-
if (!forkId) return
|
|
53
|
+
if (!forkId) return blockedDespiteAssertions('fork response carried no runId — fork annotation state not observed');
|
|
35
54
|
const list = await driver.get(`/v1/runs/${forkId}/annotations`);
|
|
36
|
-
|
|
55
|
+
expect(list.status, req('openwop.it.feedback-fork-not-copied.a-fork-of-an-annotated-run-starts-with-zero-annotations', 'RFC 0056 §B', 'GET /v1/runs/{forkId}/annotations MUST return 200 for the fork')).toBe(200);
|
|
56
|
+
const ann = (list.json as { annotations?: unknown } | undefined)?.annotations;
|
|
57
|
+
expect(Array.isArray(ann), req('openwop.it.feedback-fork-not-copied.a-fork-of-an-annotated-run-starts-with-zero-annotations', 'RFC 0056 §B', 'the annotation list MUST carry an annotations[] array')).toBe(true);
|
|
37
58
|
expect(
|
|
38
|
-
ann.length,
|
|
59
|
+
Array.isArray(ann) ? ann.length : -1,
|
|
39
60
|
req('openwop.it.feedback-fork-not-copied.a-fork-of-an-annotated-run-starts-with-zero-annotations', 'RFC 0056 §D', 'annotations are a side-store and MUST NOT be copied into a fork'),
|
|
40
61
|
).toBe(0);
|
|
41
62
|
});
|
|
@@ -52,7 +52,7 @@
|
|
|
52
52
|
*/
|
|
53
53
|
import { describe, it, expect } from 'vitest';
|
|
54
54
|
import { driver } from '../lib/driver.js';
|
|
55
|
-
import { softSkip } from '../lib/soft-skip.js';
|
|
55
|
+
import { softSkip, blockedDespiteAssertions } from '../lib/soft-skip.js';
|
|
56
56
|
import { req } from '../lib/requirement-ids.js';
|
|
57
57
|
|
|
58
58
|
const SEAM = '/v1/host/sample/test/idempotency/concurrent-claim';
|
|
@@ -133,10 +133,26 @@ describe('idempotency-concurrent-claim: two executors of one run, one effect (RF
|
|
|
133
133
|
// A host MAY cap `executors`; what it MUST NOT do is deliver more than once.
|
|
134
134
|
// Say so — an unclassified return records `blocked` with no reason, which is
|
|
135
135
|
// the same thing this file exists to stop happening to the requirement.
|
|
136
|
+
// unfailable-leg audit wave 2, 2026-09-27: the `executors: 5` probe runs after
|
|
137
|
+
// seamPresent's 200 assert, so a plain softSkip here recorded a partial-
|
|
138
|
+
// witness `executed-pass` — a host whose seam refused the in-range request
|
|
139
|
+
// (§25: range 2..8) passed without the higher-concurrency race ever running.
|
|
136
140
|
if (res.status !== 200) {
|
|
137
|
-
return
|
|
141
|
+
return blockedDespiteAssertions(`${SEAM} declined executors: 5 (status ${res.status}) — §25 accepts 2..8; the higher-concurrency race was not observed`);
|
|
138
142
|
}
|
|
139
143
|
const out = res.json as ClaimResult;
|
|
144
|
+
// unfailable-leg audit wave 2, 2026-09-27: previously the identity check below
|
|
145
|
+
// was guarded by `ids.length >= 2` and `mintedIds` defaulted to [] — a host
|
|
146
|
+
// that raced ≤2 executors, or omitted/garbled mintedIds, passed on
|
|
147
|
+
// `delivered === 1` alone, the exact vacuous pass §25 names. Now unguarded.
|
|
148
|
+
expect(
|
|
149
|
+
typeof out.attempted === 'number' && (out.attempted as number) >= 3,
|
|
150
|
+
req('openwop.it.idempotency-concurrent-claim.the-race-is-real-at-higher-concurrency-too-a-claim-that-only-holds-at-2-is-not-a', 'host-sample-test-seams.md §25', 'with executors: 5 at least three executors MUST reach the chokepoint — otherwise this leg repeats the two-executor case and witnesses nothing new'),
|
|
151
|
+
).toBe(true);
|
|
152
|
+
expect(
|
|
153
|
+
Array.isArray(out.mintedIds),
|
|
154
|
+
req('openwop.it.idempotency-concurrent-claim.the-race-is-real-at-higher-concurrency-too-a-claim-that-only-holds-at-2-is-not-a', 'host-sample-test-seams.md §25', 'the seam MUST report mintedIds, one per executor'),
|
|
155
|
+
).toBe(true);
|
|
140
156
|
expect(
|
|
141
157
|
out.delivered,
|
|
142
158
|
req('openwop.it.idempotency-concurrent-claim.the-race-is-real-at-higher-concurrency-too-a-claim-that-only-holds-at-2-is-not-a',
|
|
@@ -144,12 +160,14 @@ describe('idempotency-concurrent-claim: two executors of one run, one effect (RF
|
|
|
144
160
|
'exactly one effect regardless of how many executors race — at most one wins the compare-and-set, the rest observe the hit',
|
|
145
161
|
),
|
|
146
162
|
).toBe(1);
|
|
147
|
-
const ids =
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
163
|
+
const ids = out.mintedIds as unknown[];
|
|
164
|
+
expect(
|
|
165
|
+
ids.length,
|
|
166
|
+
req('openwop.it.idempotency-concurrent-claim.the-race-is-real-at-higher-concurrency-too-a-claim-that-only-holds-at-2-is-not-a', 'host-sample-test-seams.md §25', 'mintedIds length MUST equal attempted'),
|
|
167
|
+
).toBe(out.attempted);
|
|
168
|
+
expect(
|
|
169
|
+
new Set(ids.map((x) => JSON.stringify(x))).size,
|
|
170
|
+
req('openwop.it.idempotency-concurrent-claim.the-race-is-real-at-higher-concurrency-too-a-claim-that-only-holds-at-2-is-not-a', 'idempotency.md §"Idempotency key composition"', 'one identity across all executors at any concurrency'),
|
|
171
|
+
).toBe(1);
|
|
154
172
|
});
|
|
155
173
|
});
|
|
@@ -116,10 +116,14 @@ describe('category: core.openwop.http.idempotency-key — determinism contract',
|
|
|
116
116
|
// — the scenario soft-skips rather than failing.
|
|
117
117
|
if (!packAvailable) {
|
|
118
118
|
console.warn(`[idempotency-key-determinism] packs/core.openwop.http/index.mjs not present; skipping`);
|
|
119
|
-
|
|
119
|
+
// unfailable-leg audit wave 2, 2026-09-27: this leg asserted
|
|
120
|
+
// `packAvailable === false` inside `if (!packAvailable)` (and `=== true`
|
|
121
|
+
// in the else) — tautologies that turned a sentinel into an
|
|
122
|
+
// `executed-pass` row measuring nothing. It now records `inapplicable`
|
|
123
|
+
// with zero assertions either way.
|
|
120
124
|
return softSkip('inapplicable', 'capability or profile not advertised by this host — gate `!packAvailable` returned early ([idempotency-key-determinism] packs/core.openwop.http/index.mjs not present; skipping)');
|
|
121
125
|
}
|
|
122
|
-
|
|
126
|
+
return softSkip('inapplicable', 'sentinel leg — pack source is present; the behavioural legs in this file carry the assertions');
|
|
123
127
|
});
|
|
124
128
|
|
|
125
129
|
it('default mode (composite) — identical (runId, nodeId, payload) produces identical keys', async () => {
|
|
@@ -39,8 +39,11 @@ async function rpc(method: string, params?: Record<string, unknown>) {
|
|
|
39
39
|
}
|
|
40
40
|
|
|
41
41
|
const TEST_TOOL_NAME = `inj_${Date.now()}_${Math.random().toString(36).slice(2, 6)}`;
|
|
42
|
+
/** Set once the strict tool is registered, so the valid-args leg can tell "no tool" from "rejected". */
|
|
43
|
+
let strictToolRegistered = false;
|
|
42
44
|
|
|
43
45
|
async function registerStrictWorkflow(): Promise<boolean> {
|
|
46
|
+
if (strictToolRegistered) return true;
|
|
44
47
|
const res = await driver.post('/v1/host/sample/workflows', {
|
|
45
48
|
workflowId: `mcp.untrusted.${Date.now()}`,
|
|
46
49
|
nodes: [
|
|
@@ -60,7 +63,8 @@ async function registerStrictWorkflow(): Promise<boolean> {
|
|
|
60
63
|
},
|
|
61
64
|
],
|
|
62
65
|
});
|
|
63
|
-
|
|
66
|
+
strictToolRegistered = res.status === 200 || res.status === 201;
|
|
67
|
+
return strictToolRegistered;
|
|
64
68
|
}
|
|
65
69
|
|
|
66
70
|
describe('mcp-server-untrusted-args: advertisement shape (RFC 0020)', () => {
|
|
@@ -96,14 +100,25 @@ describe('mcp-server-untrusted-args: behavioral (RFC 0020 §D)', () => {
|
|
|
96
100
|
it('tools/call with valid arguments is accepted', async () => {
|
|
97
101
|
const cap = await readCap();
|
|
98
102
|
if (!cap || cap.supported !== true) return softSkip('inapplicable', 'capability or profile not advertised by this host — gate `!cap || cap.supported !== true` returned early');
|
|
103
|
+
// The tool this leg calls is registered by the malformed-args leg; register
|
|
104
|
+
// it here too (idempotent) so an unknown-tool error is never read as a
|
|
105
|
+
// rejection of valid arguments.
|
|
106
|
+
if (!(await registerStrictWorkflow())) return softSkip('blocked', 'precondition not met — the strict-schema tool could not be registered via /v1/host/sample/workflows');
|
|
99
107
|
const r = await rpc('tools/call', {
|
|
100
108
|
name: TEST_TOOL_NAME,
|
|
101
109
|
arguments: { text: 'hello' },
|
|
102
110
|
});
|
|
103
111
|
if (r.status === 404) return seamAbsent(`host advertises an MCP server mount but the mount (capabilities.mcp.serverUrls[0], else /v1/host/sample/mcp) answered ${r.status} — RFC 0153 §B is unobservable at the path the host itself advertised`);
|
|
104
|
-
expect(r.status).toBe(200);
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
112
|
+
expect(r.status, req('openwop.it.mcp-server-untrusted-args.tools-call-with-valid-arguments-is-accepted', 'RFC 0020 §D', 'the JSON-RPC envelope MUST answer 200')).toBe(200);
|
|
113
|
+
// unfailable-leg audit wave 2, 2026-09-27: the only requirement assert sat
|
|
114
|
+
// inside `if (r.body.error)` and checked just `code !== -32602`, so a host
|
|
115
|
+
// that rejected VALID arguments with any other JSON-RPC error (-32603,
|
|
116
|
+
// -32000, …) — or answered with neither result nor error — passed. A valid
|
|
117
|
+
// call MUST be accepted: no JSON-RPC error, and a result present.
|
|
118
|
+
// (`result.isError` is deliberately NOT asserted: it reports the exposed
|
|
119
|
+
// workflow's own execution outcome — a failed/suspended run — not a
|
|
120
|
+
// rejection of the arguments; RFC 0020 §C.)
|
|
121
|
+
expect(r.body.error, req('openwop.it.mcp-server-untrusted-args.tools-call-with-valid-arguments-is-accepted', 'RFC 0020 §D', 'valid args MUST NOT be rejected with any JSON-RPC error')).toBeUndefined();
|
|
122
|
+
expect(r.body.result, req('openwop.it.mcp-server-untrusted-args.tools-call-with-valid-arguments-is-accepted', 'RFC 0020 §D', 'an accepted tools/call MUST return a JSON-RPC result')).toBeDefined();
|
|
108
123
|
});
|
|
109
124
|
});
|
|
@@ -194,7 +194,13 @@ describe('mcp-tool-roundtrip: server wire shape', () => {
|
|
|
194
194
|
);
|
|
195
195
|
return softSkip('blocked', 'precondition not met — `!probe` returned early (seam, prior step, or fixture unavailable)');
|
|
196
196
|
}
|
|
197
|
-
|
|
197
|
+
// unfailable-leg audit wave 2, 2026-09-27: with only the in-process fake
|
|
198
|
+
// configured, this leg probed the SUITE'S OWN fake MCP server and recorded
|
|
199
|
+
// `executed-pass` — every host passed it, conforming or not, because the
|
|
200
|
+
// host under test was never contacted. The fake-path self-check is now
|
|
201
|
+
// `inapplicable` before any assertion (the host-mediated leg below is the
|
|
202
|
+
// one that exercises the host against the fake).
|
|
203
|
+
if (!probe.isReal) return softSkip('inapplicable', 'suite fixture self-test — host not exercised (only the in-process MCP fake is configured; set OPENWOP_MCP_REAL_SERVER_URL for real-server interop evidence)');
|
|
198
204
|
|
|
199
205
|
// Per MCP `initialize` spec, params MUST carry protocolVersion +
|
|
200
206
|
// capabilities + clientInfo. The in-process fake accepts empty
|
|
@@ -258,26 +264,6 @@ describe('mcp-tool-roundtrip: server wire shape', () => {
|
|
|
258
264
|
`[mcp-tool-roundtrip] real-server interop OK against ${probe.url} ` +
|
|
259
265
|
`(tool=${first?.name}, isError=${callResult.isError === true})`,
|
|
260
266
|
);
|
|
261
|
-
} else {
|
|
262
|
-
// Fake-server path: deterministic echo tool, assert verbatim.
|
|
263
|
-
expect(listResult.tools?.some((t) => t.name === 'echo')).toBe(true);
|
|
264
|
-
|
|
265
|
-
const call = await postJsonRpc(
|
|
266
|
-
probe.url,
|
|
267
|
-
'tools/call',
|
|
268
|
-
{ name: 'echo', arguments: { text: 'hello-from-conformance' } },
|
|
269
|
-
3,
|
|
270
|
-
);
|
|
271
|
-
expect(call.status).toBe(200);
|
|
272
|
-
const callResult = (call.json.result ?? {}) as {
|
|
273
|
-
content?: ReadonlyArray<{ type?: string; text?: string }>;
|
|
274
|
-
};
|
|
275
|
-
expect(callResult.content?.[0]?.type).toBe('text');
|
|
276
|
-
expect(callResult.content?.[0]?.text).toBe('hello-from-conformance');
|
|
277
|
-
|
|
278
|
-
const fake = getMcpFakeServer()!;
|
|
279
|
-
const methods = fake.invocations().map((i) => i.method);
|
|
280
|
-
expect(methods).toEqual(['initialize', 'tools/list', 'tools/call']);
|
|
281
267
|
}
|
|
282
268
|
});
|
|
283
269
|
});
|