@openwop/openwop-conformance 2.42.0 → 2.42.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/README.md +1 -1
- package/dist/spec-artifacts.lock.json +2 -2
- package/package.json +2 -2
- package/requirements.json +9 -8
- package/schemas/CORPUS-STAMP.json +10 -10
- package/src/lib/polling.ts +21 -0
- package/src/scenarios/context-budget-transcript-bound.test.ts +3 -3
- package/src/scenarios/context-summarization-replay.test.ts +4 -4
- package/src/scenarios/v2-mcp-mount-map.test.ts +20 -4
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# `@openwop/openwop-conformance` Changelog
|
|
2
2
|
|
|
3
|
+
## [2.42.1] — 2026-09-26 — RFC 0111's live witness scenarios outlast a real host, and the MCP run-transport leg identifies its own run
|
|
4
|
+
|
|
5
|
+
- **RFC 0111's live scenarios no longer die at vitest's 30 s default.** `context-budget-transcript-bound` and `context-summarization-replay` drive the live-model fixture `conformance-context-budget-live` (six real model turns and six child runs; 30–40 s per run on MyndHyve production, 2026-09-26), and the global `testTimeout: 30_000` killed both before any assertion — which `OPENWOP_POLL_TIMEOUT_SCALE` could not reach. Each now carries a per-test timeout from `liveScenarioTimeoutMs(runs)` (`src/lib/polling.ts`: the scaled sum of `LIVE_RUN_POLL_MS` = 180 s per run plus 60 s for seam reads — 240 s / 420 s at scale 1) and polls with `{ timeoutMs: LIVE_RUN_POLL_MS }`, so the named poll deadline always fires before the test deadline, at every scale. The global timeout is unchanged. Self-test: `src/lib/polling.test.ts` (3 new); an unscaled helper fails 2 of them and a 30 s helper 3.
|
|
6
|
+
- **`v2-mcp-mount-map`'s run-transport leg identifies its own run.** It diffed `listRuns` around one `tools/call` and required exactly one new `conformance-noop` run, but the suite runs files concurrently and that fixture is every file's smallest run: the v2 reference host's CI (4 workers) failed "got 2 new run(s)" on a host whose `tools/call` started exactly one, and `fresh[0]` could have been a sibling's run read for the wrong transport. The leg now requires at least one new run and reads the transport of the run the result names (when `listRuns` shows it), else the only new run, else passes if any new run in the window started `mcp`.
|
|
7
|
+
- **`spec/v2/core/webhooks.md`'s `Stable` banner cites RFC 0217**, now `Accepted` on the v2 reference host's certified 2.42.0 cut. No scenario change.
|
|
8
|
+
|
|
3
9
|
## [2.42.0] — 2026-09-26 — `replay_context_summary_unavailable` is a registered code, and an unregistered subscription has no sink
|
|
4
10
|
|
|
5
11
|
- **RFC 0217: after unregister, the dead-letter read answers `404 not_found`, as for a subscription that never existed.** `v2-webhook-durable-delivery` gains `openwop.requirement.0217.dead-letter-read-after-unregister`, gated on `webhooks.deadLetter`: read the sink (`200`, the control), unregister (`204`), read again, and read a never-minted same-tenant id; both later reads must be `404 not_found`. Closes RFC 0215's gap G5. Sabotage: a v2 reference host whose read answers `200 { deliveries: [] }` for an unknown id fails the row.
|
package/README.md
CHANGED
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
# --legacy-peer-deps is REQUIRED, not optional: the exact peer pin is what npm's
|
|
12
12
|
# default resolver refuses. npm 10.9 fails outright with
|
|
13
13
|
# "Cannot read properties of null (reading 'edgesOut')" — use npm >= 11.
|
|
14
|
-
npm install --legacy-peer-deps @openwop/openwop-conformance@2.42.
|
|
14
|
+
npm install --legacy-peer-deps @openwop/openwop-conformance@2.42.1 @openwop/spec-artifacts@2.42.1
|
|
15
15
|
# or run without install:
|
|
16
16
|
npx @openwop/openwop-conformance --base-url https://api.example.com --api-key hk_test_...
|
|
17
17
|
```
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@openwop/openwop-conformance",
|
|
3
|
-
"version": "2.42.
|
|
3
|
+
"version": "2.42.1",
|
|
4
4
|
"description": "Production-ready black-box conformance suite for OpenWOP v1.0 compliant servers.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -56,6 +56,6 @@
|
|
|
56
56
|
"@openwop/spec-artifacts": "file:../spec-artifacts"
|
|
57
57
|
},
|
|
58
58
|
"peerDependencies": {
|
|
59
|
-
"@openwop/spec-artifacts": "2.42.
|
|
59
|
+
"@openwop/spec-artifacts": "2.42.1"
|
|
60
60
|
}
|
|
61
61
|
}
|
package/requirements.json
CHANGED
|
@@ -30273,14 +30273,15 @@
|
|
|
30273
30273
|
},
|
|
30274
30274
|
{
|
|
30275
30275
|
"section": "interop-map.json mcp.methods tools/call; runs.md run.started",
|
|
30276
|
-
"requirement":
|
|
30276
|
+
"requirement": null,
|
|
30277
|
+
"interpolated": true
|
|
30277
30278
|
}
|
|
30278
30279
|
]
|
|
30279
30280
|
},
|
|
30280
30281
|
{
|
|
30281
30282
|
"id": "openwop.it.v2-mcp-mount-map.a-suspending-tool-answers-inputrequiredresult-the-retry-resolves-it-requeststate",
|
|
30282
30283
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30283
|
-
"line":
|
|
30284
|
+
"line": 197,
|
|
30284
30285
|
"title": "a suspending tool answers InputRequiredResult; the retry resolves it; requestState is single use and forgery-proof",
|
|
30285
30286
|
"explicitId": null,
|
|
30286
30287
|
"citations": [
|
|
@@ -30322,7 +30323,7 @@
|
|
|
30322
30323
|
{
|
|
30323
30324
|
"id": "openwop.it.v2-mcp-mount-map.the-mrtr-input-request-key-is-the-open-interrupt-s-interruptid-never-the-node-it",
|
|
30324
30325
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30325
|
-
"line":
|
|
30326
|
+
"line": 222,
|
|
30326
30327
|
"title": "the MRTR input-request key is the open interrupt’s interruptId, never the node it suspended on",
|
|
30327
30328
|
"explicitId": null,
|
|
30328
30329
|
"citations": [
|
|
@@ -30361,7 +30362,7 @@
|
|
|
30361
30362
|
{
|
|
30362
30363
|
"id": "openwop.it.v2-mcp-mount-map.a-list-that-differs-per-caller-is-cachescope-private",
|
|
30363
30364
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30364
|
-
"line":
|
|
30365
|
+
"line": 260,
|
|
30365
30366
|
"title": "a list that differs per caller is cacheScope private",
|
|
30366
30367
|
"explicitId": null,
|
|
30367
30368
|
"citations": [
|
|
@@ -30379,7 +30380,7 @@
|
|
|
30379
30380
|
{
|
|
30380
30381
|
"id": "openwop.it.v2-mcp-mount-map.unknown-meta-extension-keys-and-capabilities-extensions-are-opaque-processed-nor",
|
|
30381
30382
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30382
|
-
"line":
|
|
30383
|
+
"line": 277,
|
|
30383
30384
|
"title": "unknown _meta extension keys and capabilities.extensions are opaque: processed normally, granting nothing",
|
|
30384
30385
|
"explicitId": null,
|
|
30385
30386
|
"citations": [
|
|
@@ -30397,7 +30398,7 @@
|
|
|
30397
30398
|
{
|
|
30398
30399
|
"id": "openwop.it.v2-mcp-mount-map.an-unauthenticated-request-is-refused-at-the-boundary-unless-anonymousactor-is-a",
|
|
30399
30400
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30400
|
-
"line":
|
|
30401
|
+
"line": 286,
|
|
30401
30402
|
"title": "an unauthenticated request is refused at the boundary unless anonymousActor is advertised",
|
|
30402
30403
|
"explicitId": null,
|
|
30403
30404
|
"citations": [
|
|
@@ -30420,7 +30421,7 @@
|
|
|
30420
30421
|
{
|
|
30421
30422
|
"id": "openwop.it.v2-mcp-mount-map.a-credential-interrupt-is-answered-in-url-mode-url-connecturl-iserror-without-ur",
|
|
30422
30423
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30423
|
-
"line":
|
|
30424
|
+
"line": 303,
|
|
30424
30425
|
"title": "a credential interrupt is answered in URL mode (url = connectUrl), isError without URL support, and input_required again on an accept retry with no credential",
|
|
30425
30426
|
"explicitId": "openwop.requirement.0199.mcp-url-mode",
|
|
30426
30427
|
"citations": [
|
|
@@ -30458,7 +30459,7 @@
|
|
|
30458
30459
|
{
|
|
30459
30460
|
"id": "openwop.it.v2-mcp-mount-map.form-mode-is-never-emitted-for-a-nested-schema-or-a-sensitive-format-password-fi",
|
|
30460
30461
|
"file": "v2-mcp-mount-map.test.ts",
|
|
30461
|
-
"line":
|
|
30462
|
+
"line": 344,
|
|
30462
30463
|
"title": "form mode is never emitted for a nested schema or a sensitive (format password) field",
|
|
30463
30464
|
"explicitId": "openwop.requirement.0199.form-mode-no-secret",
|
|
30464
30465
|
"citations": [
|
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
{
|
|
2
2
|
"_comment": "Provenance of @openwop/spec-artifacts (RFC 0168 §D.2). files: SHA-256 per file; the conformance suite compares the installed peer against dist/spec-artifacts.lock.json at start.",
|
|
3
3
|
"package": "@openwop/spec-artifacts",
|
|
4
|
-
"version": "2.42.
|
|
5
|
-
"corpusTag": "v2.42.
|
|
4
|
+
"version": "2.42.1",
|
|
5
|
+
"corpusTag": "v2.42.1",
|
|
6
6
|
"files": {
|
|
7
7
|
"api/.redocly.lint-ignore.yaml": "bf5a8350b88a72fa43f59605ed8d903ed24b6cfccda5e45509c9f6ed9ee4e712",
|
|
8
8
|
"api/asyncapi.yaml": "d5ecb9ee6114582be3b1f662c84bfac9ae96dae7bacb853e461168f70a8e1c7d",
|
|
9
9
|
"api/grpc/openwop.proto": "c3e72bb17cba514ee98feb6434e6c9b6ea6795bfd086489ec69fd882dd1ad977",
|
|
10
10
|
"api/openapi.yaml": "6366b5514be71098e9ea41fc81ff52bc39db1a547b353c0b344cc2a2bf5370aa",
|
|
11
11
|
"api/redocly.yaml": "b0604c89b2ca6d5076ec25725c539dad44a741a811fe524439ee6daef8baa09f",
|
|
12
|
-
"api/seams-v2.yaml": "
|
|
13
|
-
"api/v2/asyncapi.yaml": "
|
|
14
|
-
"api/v2/openapi.yaml": "
|
|
12
|
+
"api/seams-v2.yaml": "849b6c3d37336132343468e553b413a61a133fbc943d21ffc8e961948284db9f",
|
|
13
|
+
"api/v2/asyncapi.yaml": "96bac052de733a0987c35b4834341d91633381ff2ae1e6087ddb2956ac1edad9",
|
|
14
|
+
"api/v2/openapi.yaml": "eb39b2957737fc0f2e3187c33aac2f1475a3d2aa382a5c57686a1a15a5b643d4",
|
|
15
15
|
"api/v2/redocly.yaml": "1e66b60e6118ad11a823bb620678be464d99dfe50a40e3e6f93ec9429b88b34c",
|
|
16
16
|
"schemas/README.md": "0c0b737ffcf8f30e7d2809cec8a498232de710f41443212922ad8337cdde0b51",
|
|
17
17
|
"schemas/a2a-task-state.schema.json": "75d5049dea8bd873ff0e7546f1c60c8d36c219bec7084264360be54a8be30ae1",
|
|
@@ -204,7 +204,7 @@
|
|
|
204
204
|
"schemas/workspace-file.schema.json": "464de85c2a068243084ee9c1d969bc7cd5d8f7948574e58450d6493c38a0e1e4",
|
|
205
205
|
"spec/v1/alias-detectors.json": "2401fcb1c18cdd688c018b3d716ae6bca85c5e356220ee2b793bd9d81872412d",
|
|
206
206
|
"spec/v1/capability-declaration-classes.json": "e7729aed5c4b4e1dd02abab0530f14cc95f5d4070fe51fb139e7f5cccefa00c6",
|
|
207
|
-
"spec/v1/core-standard-manifest.json": "
|
|
207
|
+
"spec/v1/core-standard-manifest.json": "bbcefda69bffa9fb67803c48ddc7e786866c655d8af36f73fc0e3b27e69a1917",
|
|
208
208
|
"spec/v1/deprecations.json": "2d03f4729810280147ea08c630ea65434f2dc5370567d0aac41c3595f8337d40",
|
|
209
209
|
"spec/v1/deprecations.schema.json": "3e393c405d2a41b467d8c5e3c468549078df1ce6a6d2588fc488097b95d9b55b",
|
|
210
210
|
"spec/v1/event-codemap.json": "3da60d884157793a360da532a9fcbbfb5285636db325a74cec94b34622186d97",
|
|
@@ -240,7 +240,7 @@
|
|
|
240
240
|
"spec/v2/core/security-defaults.md": "500471a8db7af9b776ac40b4a9a278c9f3d45e1c1dead880114e7fadc7bee37d",
|
|
241
241
|
"spec/v2/core/tool-catalog.md": "35f1a3fd509db0fc0490ddf400686fe94dc4b20c98064d5ebe776da516153e60",
|
|
242
242
|
"spec/v2/core/versioning.md": "0aded71d090c8358c3ce17763c120cfc1cc42531a5eaa1b8a2a7016ed9594069",
|
|
243
|
-
"spec/v2/core/webhooks.md": "
|
|
243
|
+
"spec/v2/core/webhooks.md": "78d7e237866bf1ee95bbfa9d5b62b619ca2f8ec677d1b752ba96ed5efcfc77ef",
|
|
244
244
|
"spec/v2/core/workflow-chain-packs.md": "ee45d3fede3bb6f0cc6abcf0d6c929c7cd213a0fbe862b738a998f929c4d1102",
|
|
245
245
|
"spec/v2/corrections.json": "cc74ffde74384b0f261a4f625871d922e8661d654d647a8534e21036595cb280",
|
|
246
246
|
"spec/v2/corrections.schema.json": "4ac595b9a6d7f66d53de03fe0edbe0dc58de46426932387bfdc0ceb3cf949a3e",
|
|
@@ -294,9 +294,9 @@
|
|
|
294
294
|
"spec/v2/path-manifest.json": "ae9b56d03a701061fd48ad0b73cb0373493f6a4547da92763b422295b5e19535",
|
|
295
295
|
"spec/v2/peer-dependency-aliases.json": "d10299280abee08258502925bc327293ee413e0108cd6e6ec75ff6110653308d",
|
|
296
296
|
"spec/v2/profiles.json": "0636f19fceae625390003a347e70ef4797d84766b5c24ce8a02cea52aadebca4",
|
|
297
|
-
"spec/v2/release.json": "
|
|
297
|
+
"spec/v2/release.json": "c226854124edd701082c57b55c908cdf09926837611a59d00e306696a06a9b8a",
|
|
298
298
|
"spec/v2/retention-floors.json": "eaf3722d95c79947af1d4269ef85117e126518c588cfcf1a2b21b97269f51624",
|
|
299
|
-
"spec/v2/surface-baseline.json": "
|
|
299
|
+
"spec/v2/surface-baseline.json": "8411f06cd9888524b6d234e8d111896d8c90d06991453c9a192e7550d0b3cb9e"
|
|
300
300
|
},
|
|
301
|
-
"corpusCommit": "
|
|
301
|
+
"corpusCommit": "c92bac3ebafda0fab5a6113df67422efe428d70d"
|
|
302
302
|
}
|
package/src/lib/polling.ts
CHANGED
|
@@ -71,6 +71,27 @@ export function scaledTimeoutMs(timeoutMs: number): number {
|
|
|
71
71
|
return scale === 1 ? timeoutMs : Math.ceil(timeoutMs * scale);
|
|
72
72
|
}
|
|
73
73
|
|
|
74
|
+
/**
|
|
75
|
+
* Live-model scenarios (RFC 0111's `conformance-context-budget-live`) drive real
|
|
76
|
+
* model turns and child runs; one run takes 30–40 s on a production host
|
|
77
|
+
* (MyndHyve, 2026-09-26), so vitest's global 30 s `testTimeout` killed them
|
|
78
|
+
* before any assertion — a suite defect that looks like a host failure.
|
|
79
|
+
*
|
|
80
|
+
* `LIVE_RUN_POLL_MS` is the base bound for ONE live run to reach a terminal
|
|
81
|
+
* status (below the fixture's own `settings.timeout` of 300 s); pass it as
|
|
82
|
+
* `pollUntilTerminal(runId, { timeoutMs: LIVE_RUN_POLL_MS })` — `pollUntil`
|
|
83
|
+
* scales it. `liveScenarioTimeoutMs(runs)` is the per-test vitest timeout: the
|
|
84
|
+
* scaled sum of the scenario's poll bounds plus 60 s for its seam reads, so the
|
|
85
|
+
* poll deadline always fires first and a hung host fails with a named poll
|
|
86
|
+
* message, never a bare vitest timeout. Both scale with
|
|
87
|
+
* `OPENWOP_POLL_TIMEOUT_SCALE`, so their order holds at every scale.
|
|
88
|
+
*/
|
|
89
|
+
export const LIVE_RUN_POLL_MS = 180_000;
|
|
90
|
+
|
|
91
|
+
export function liveScenarioTimeoutMs(runs: number): number {
|
|
92
|
+
return scaledTimeoutMs(runs * LIVE_RUN_POLL_MS + 60_000);
|
|
93
|
+
}
|
|
94
|
+
|
|
74
95
|
const TERMINAL = new Set(['completed', 'failed', 'cancelled']);
|
|
75
96
|
|
|
76
97
|
export async function getRun(runId: string): Promise<RunSnapshot> {
|
|
@@ -55,7 +55,7 @@
|
|
|
55
55
|
|
|
56
56
|
import { describe, it, expect } from 'vitest';
|
|
57
57
|
import { driver } from '../lib/driver.js';
|
|
58
|
-
import { pollUntilTerminal } from '../lib/polling.js';
|
|
58
|
+
import { LIVE_RUN_POLL_MS, liveScenarioTimeoutMs, pollUntilTerminal } from '../lib/polling.js';
|
|
59
59
|
import { behaviorGate } from '../lib/behavior-gate.js';
|
|
60
60
|
import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
61
61
|
import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
|
|
@@ -111,7 +111,7 @@ describe('context-budget-transcript-bound (RFC 0111 §"Context economy")', () =>
|
|
|
111
111
|
const runId = runIdOf(create.json);
|
|
112
112
|
expect(runId, req(ID, 'RFC 0111', 'the create response MUST carry a runId')).toBeDefined();
|
|
113
113
|
if (runId === undefined) return softSkip('blocked', 'no runId');
|
|
114
|
-
await pollUntilTerminal(runId);
|
|
114
|
+
await pollUntilTerminal(runId, { timeoutMs: LIVE_RUN_POLL_MS });
|
|
115
115
|
|
|
116
116
|
const windows: Array<{ iteration: number; window: TranscriptWindow }> = [];
|
|
117
117
|
for (let iteration = 1; iteration <= MAX_ITERATIONS_PROBED; iteration += 1) {
|
|
@@ -153,5 +153,5 @@ describe('context-budget-transcript-bound (RFC 0111 §"Context economy")', () =>
|
|
|
153
153
|
|
|
154
154
|
if (log === null) return softSkip('blocked', 'the run event-log seam is unavailable, so the real-event, recent-tail and pressure rules were not measured');
|
|
155
155
|
if (!pressure) softSkip('inapplicable', `no iteration shows budget pressure — every eligible event fit under transcriptTokenBudget ${budget}, so the bound was never exercised (a budget the run never reaches is not a witness)`);
|
|
156
|
-
});
|
|
156
|
+
}, liveScenarioTimeoutMs(1));
|
|
157
157
|
});
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
|
|
39
39
|
import { describe, it, expect } from 'vitest';
|
|
40
40
|
import { driver } from '../lib/driver.js';
|
|
41
|
-
import { pollUntilTerminal } from '../lib/polling.js';
|
|
41
|
+
import { LIVE_RUN_POLL_MS, liveScenarioTimeoutMs, pollUntilTerminal } from '../lib/polling.js';
|
|
42
42
|
import { behaviorGate } from '../lib/behavior-gate.js';
|
|
43
43
|
import { isFixtureAdvertised } from '../lib/fixtures.js';
|
|
44
44
|
import { readCapabilityFamily } from '../lib/discovery-capabilities.js';
|
|
@@ -125,7 +125,7 @@ describe('context-summarization-replay (RFC 0111 §"Replay determinism")', () =>
|
|
|
125
125
|
const sourceRunId = runIdOf(create.json);
|
|
126
126
|
expect(sourceRunId, req(ID, 'rest-endpoints.md POST /v1/runs', 'the create response MUST carry a runId')).toBeDefined();
|
|
127
127
|
if (sourceRunId === undefined) return softSkip('blocked', 'no runId');
|
|
128
|
-
await pollUntilTerminal(sourceRunId);
|
|
128
|
+
await pollUntilTerminal(sourceRunId, { timeoutMs: LIVE_RUN_POLL_MS });
|
|
129
129
|
|
|
130
130
|
const sourceQ = await queryTestEvents(sourceRunId);
|
|
131
131
|
if (!sourceQ.ok) return softSkip('blocked', 'the run event-log seam is unavailable');
|
|
@@ -141,7 +141,7 @@ describe('context-summarization-replay (RFC 0111 §"Replay determinism")', () =>
|
|
|
141
141
|
const forkRunId = runIdOf(fork.json);
|
|
142
142
|
expect(forkRunId, req(ID, 'rest-endpoints.md POST /v1/runs/{runId}:fork', 'replay fork MUST return a runId')).toBeDefined();
|
|
143
143
|
if (forkRunId === undefined) return softSkip('blocked', 'no fork runId');
|
|
144
|
-
await pollUntilTerminal(forkRunId);
|
|
144
|
+
await pollUntilTerminal(forkRunId, { timeoutMs: LIVE_RUN_POLL_MS });
|
|
145
145
|
|
|
146
146
|
const forkQ = await queryTestEvents(forkRunId);
|
|
147
147
|
if (!forkQ.ok) return softSkip('blocked', 'the event-log seam is unavailable for the fork');
|
|
@@ -153,5 +153,5 @@ describe('context-summarization-replay (RFC 0111 §"Replay determinism")', () =>
|
|
|
153
153
|
if (sourceTexts === null || forkTexts === null) return softSkip('inapplicable', 'the transcript-window seam serves no entries[] for these runs, so the model-facing summary text was not compared (summaryRef reuse was)');
|
|
154
154
|
expect(sourceTexts.length, req(ID, 'RFC 0111 §"Replay determinism"', 'the source run summarized, so its transcript windows MUST carry the summary text it fed')).toBeGreaterThan(0);
|
|
155
155
|
expect(forkTexts, req(ID, 'RFC 0111 §"Replay determinism"', 'the replay MUST feed the model the recorded summary text, byte for byte — never a re-summarization')).toEqual(sourceTexts);
|
|
156
|
-
});
|
|
156
|
+
}, liveScenarioTimeoutMs(2));
|
|
157
157
|
});
|
|
@@ -172,10 +172,26 @@ describe('RFC 0208 — v2-mcp-mount-map (host as MCP 2026-07-28 server, gated on
|
|
|
172
172
|
const before = new Set(await list());
|
|
173
173
|
const ok = await toolCall(m.url, 'conformance-noop');
|
|
174
174
|
const fresh = (await list()).filter((x) => !before.has(x));
|
|
175
|
-
expect(fresh.length, req(R('mcp-run-transport'), 'interop-map.json mcp.methods tools/call', `tools/call MUST start a run (v2Operation createRun) that listRuns shows the same Subject (got ${fresh.length} new run(s); result ${JSON.stringify(ok.error ?? ok.result)})`)).
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
175
|
+
expect(fresh.length, req(R('mcp-run-transport'), 'interop-map.json mcp.methods tools/call', `tools/call MUST start a run (v2Operation createRun) that listRuns shows the same Subject (got ${fresh.length} new run(s); result ${JSON.stringify(ok.error ?? ok.result)})`)).toBeGreaterThanOrEqual(1);
|
|
176
|
+
// WHICH new run is ours (2.42.1). The leg counted `fresh.length === 1`, but
|
|
177
|
+
// the suite runs files concurrently and conformance-noop is every file's
|
|
178
|
+
// smallest run, so a sibling's run landed in the same window: the v2
|
|
179
|
+
// reference host's CI (4 workers) failed "got 2 new run(s)" on a host whose
|
|
180
|
+
// own tools/call started exactly one, and `fresh[0]` could have been the
|
|
181
|
+
// sibling's run, read for the wrong transport. Ours is the runId the result
|
|
182
|
+
// names when it names one listRuns shows, else the only new run; when the
|
|
183
|
+
// window stays ambiguous, the requirement holds if any new run started mcp.
|
|
184
|
+
const named = ((): string | null => {
|
|
185
|
+
const m = /"runId"\s*:\s*"([^"]+)"/.exec(JSON.stringify(ok.result ?? ''));
|
|
186
|
+
return m && fresh.includes(m[1]!) ? m[1]! : null;
|
|
187
|
+
})();
|
|
188
|
+
const transportOf = async (runId: string): Promise<string | undefined> => {
|
|
189
|
+
const poll = await driver.get(`/runs/${encodeURIComponent(runId)}/events/poll?timeout=1`);
|
|
190
|
+
return ((poll.json as { events?: Array<{ type?: string; payload?: { transport?: string } }> } | undefined)?.events ?? []).find((e) => e.type === 'run.started')?.payload?.transport;
|
|
191
|
+
};
|
|
192
|
+
const ours = named ?? (fresh.length === 1 ? fresh[0]! : null);
|
|
193
|
+
const transports = ours !== null ? [await transportOf(ours)] : await Promise.all(fresh.map(transportOf));
|
|
194
|
+
expect(transports.includes('mcp') ? 'mcp' : transports[0], req(R('mcp-run-transport'), 'interop-map.json mcp.methods tools/call; runs.md run.started', `the run starts with run.started.transport mcp (${ours !== null ? `run ${ours}` : `${fresh.length} runs started in the window, none named by the result`}; transports ${JSON.stringify(transports)})`)).toBe('mcp');
|
|
179
195
|
});
|
|
180
196
|
|
|
181
197
|
it('a suspending tool answers InputRequiredResult; the retry resolves it; requestState is single use and forgery-proof', async () => {
|