@yanlinglabs/winter-conformance 0.0.14 → 0.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/README.md +2 -2
  2. package/dist/official/differential-harness.d.ts +90 -0
  3. package/dist/official/task-frames-script.d.ts +95 -0
  4. package/goldens/advertised-set-round.trace.json +9 -3
  5. package/goldens/background-task-round.trace.json +7 -1
  6. package/goldens/bash-background-round.trace.json +11 -3
  7. package/goldens/canusetool-approved-round.trace.json +7 -1
  8. package/goldens/compaction-auto-round.trace.json +7 -1
  9. package/goldens/compaction-manual-round.trace.json +7 -1
  10. package/goldens/denied-tool-round.trace.json +7 -1
  11. package/goldens/hook-denied-round.trace.json +7 -1
  12. package/goldens/hooked-tool-round.trace.json +7 -1
  13. package/goldens/interrupt.trace.json +7 -1
  14. package/goldens/mcp-tool-round.trace.json +6 -0
  15. package/goldens/messaging-facet-round.trace.json +7 -1
  16. package/goldens/mode-switch-mid-session.trace.json +7 -1
  17. package/goldens/multi-turn.trace.json +11 -5
  18. package/goldens/p6-anthropic-fake.trace.json +9 -2
  19. package/goldens/p6-gemini-fake.trace.json +9 -2
  20. package/goldens/p6-openai-chat-fake.trace.json +9 -2
  21. package/goldens/p6-openai-responses-fake.trace.json +9 -2
  22. package/goldens/p6-resolution-failure.trace.json +7 -1
  23. package/goldens/plain-query.trace.json +9 -3
  24. package/goldens/resume.trace.json +18 -6
  25. package/goldens/sendmessage-child-round.trace.json +63 -7
  26. package/goldens/skill-invocation-round.trace.json +7 -1
  27. package/goldens/structured-exhaustion-round.trace.json +7 -1
  28. package/goldens/structured-output-round.trace.json +7 -1
  29. package/goldens/subagent-permission-round.trace.json +84 -10
  30. package/goldens/subagent-spawn-round.trace.json +61 -5
  31. package/goldens/tool-round.trace.json +7 -1
  32. package/goldens/toolsearch-select-round.trace.json +6 -0
  33. package/package.json +1 -1
package/README.md CHANGED
@@ -2,8 +2,8 @@
2
2
 
3
3
  Winter's SDK compatibility corpus: a trace normalizer, a set of committed golden traces, and the
4
4
  pinned-upstream ("official SDK") mechanics that back Winter's compatibility claim against
5
- `@anthropic-ai/claude-agent-sdk@0.3.250` (see [WS-02](../../../docs/superpowers/specs/winter/WS-02-repo-and-packaging.md)
6
- in the `winter-agent-sdk` repository for the full spec, if you have it checked out).
5
+ `@anthropic-ai/claude-agent-sdk@0.3.250`. See this package's own `src/` layout and the
6
+ `winter-agent-sdk` repository's packaging conventions for how the corpus is versioned and published.
7
7
 
8
8
  This package is published to GitHub Packages under restricted access (`@yanlinglabs` scope). The
9
9
  registry is chosen by the release workflow, not by a committed pin — see
@@ -0,0 +1,90 @@
1
+ export declare const CLAUDE_VERSION = "0.3.250";
2
+ export declare const OFFICIAL_MODEL = "claude-haiku-4-5";
3
+ export type RawFrame = Record<string, unknown>;
4
+ export type ResolvedBinary = {
5
+ binaryPath: string;
6
+ } | {
7
+ reason: string;
8
+ };
9
+ /**
10
+ * Resolves (fetching + installing on first use, into a CACHED prefix reused across runs) the pinned
11
+ * claude binary. Returns `{reason}` -- never throws -- for every reason it might be unavailable, so
12
+ * every caller's `describe.skipIf` can report the reason plainly.
13
+ */
14
+ export declare function resolvePinnedClaudeBinary(): Promise<ResolvedBinary>;
15
+ export interface OfficialRoots {
16
+ root: string;
17
+ home: string;
18
+ cfg: string;
19
+ cwd: string;
20
+ }
21
+ /** Fresh mkdtemp HOME/CLAUDE_CONFIG_DIR/cwd under one owned root -- never the real `~/.claude`. */
22
+ export declare function makeOfficialRoots(prefix: string): OfficialRoots;
23
+ export declare function cleanupRoots(r: OfficialRoots): void;
24
+ /**
25
+ * The explicit, minimal env every spawn in this family uses -- never a `process.env` spread. `extra`
26
+ * merges last (an env var a specific scenario needs, e.g. `CLAUDE_CODE_FORK_SUBAGENT`).
27
+ */
28
+ export declare function minimalOfficialEnv(opts: {
29
+ home: string;
30
+ cfg: string;
31
+ baseUrl: string;
32
+ extra?: Record<string, string>;
33
+ }): Record<string, string>;
34
+ export interface SseEvent {
35
+ event: string;
36
+ data: unknown;
37
+ }
38
+ export declare function sseResponse(events: SseEvent[]): Response;
39
+ /** One assistant turn carrying N `tool_use` blocks (batched, as a model that calls several tools at
40
+ * once does -- scenario 4's two sibling forks are one assistant message with two blocks). */
41
+ export declare function sseToolUseTurn(blocks: Array<{
42
+ id: string;
43
+ name: string;
44
+ input: unknown;
45
+ }>, opts?: {
46
+ msgId?: string;
47
+ }): SseEvent[];
48
+ export declare function sseTextTurn(text: string, opts?: {
49
+ msgId?: string;
50
+ }): SseEvent[];
51
+ /**
52
+ * A `Bun.serve` fake of the Anthropic Messages API that keeps the RAW request body (never just
53
+ * `.messages`, unlike `task-frames-differential.test.ts`'s own inline server -- request-layout
54
+ * scenarios need `system`/`tools`/every top-level key too) and routes every POST by content through
55
+ * the caller's `route` callback. Every request is logged to stderr (structural facts only -- request
56
+ * count, path, message count/roles -- never the system/tools prose, matching this whole file
57
+ * family's own logging discipline). Non-POST hits (health-check pings) get a benign ack.
58
+ */
59
+ export declare function startCapturingLoopback(route: (messages: RawFrame[], body: RawFrame, count: number) => Response, logPrefix?: string): {
60
+ url: string;
61
+ requests: RawFrame[];
62
+ stop: () => void;
63
+ };
64
+ export interface OfficialStreamSession {
65
+ /** Writes one NDJSON `type:"user"` frame to stdin and flushes -- the exact shape the pinned binary's `--input-format stream-json` accepts, verified against a real spawn (spike, 2026-09-17). */
66
+ send(text: string): void;
67
+ /** Reads (and buffers) stdout frames until `pred` matches one (inclusive); returns every frame seen so far, from the start of the stream. Throws with the stderr tail on timeout. */
68
+ readUntil(pred: (frame: RawFrame) => boolean, timeoutMs?: number): Promise<RawFrame[]>;
69
+ /** Waits `graceMs`, returning whatever NEW frames (since the last `readUntil`/`drainQuiet` call) arrived in that window -- used to prove nothing MORE arrives unprompted (P16-3's held-vs-immediate `result` claim). */
70
+ drainQuiet(graceMs: number): Promise<RawFrame[]>;
71
+ /** Every frame seen so far, in order (same backing array `readUntil`/`drainQuiet` read from). */
72
+ allFrames: RawFrame[];
73
+ /** Ends stdin -- the binary's own `--input-format stream-json` exits once it sees EOF and no task is holding it open. */
74
+ close(): void;
75
+ exited: Promise<number | null>;
76
+ stderrTail(n?: number): string;
77
+ }
78
+ /**
79
+ * Spawns the pinned binary with stdin held open (`-p --input-format stream-json --output-format
80
+ * stream-json --verbose`), the shape scenarios 1/2/5 need (a two-turn or open-ended conversation in
81
+ * ONE process, matching R3a's "streaming-input hosts... never get a held result" claim and R4's
82
+ * "index-0 stable across turns" claim -- both need turns inside a single live session, not two
83
+ * separate `--resume` invocations). `opts.args` are extra flags appended after the fixed base set.
84
+ */
85
+ export declare function spawnOfficialStreamJson(opts: {
86
+ binaryPath: string;
87
+ env: Record<string, string>;
88
+ cwd: string;
89
+ args?: string[];
90
+ }): OfficialStreamSession;
@@ -0,0 +1,95 @@
1
+ export declare const TOOL_USE_BG = "toolu_bg1";
2
+ export declare const TOOL_USE_FG = "toolu_fg2";
3
+ export declare const TOOL_USE_AGENT = "toolu_agent3";
4
+ export declare const TOOL_USE_CHILD_ECHO = "toolu_child_echo";
5
+ export declare const BG_COMMAND = "sleep 1; exit 3";
6
+ export declare const FG_COMMAND = "sleep 3";
7
+ export declare const BG_DESCRIPTION = "bg fail";
8
+ export declare const FG_DESCRIPTION = "fg sleep";
9
+ export declare const AGENT_DESCRIPTION = "child probe";
10
+ export declare const CHILD_PROMPT = "run echo";
11
+ export declare const SUBAGENT_TYPE = "general-purpose";
12
+ export declare const CHILD_ECHO_COMMAND = "echo hi";
13
+ export declare const PARENT_FINAL_TEXT = "parent: all done";
14
+ export declare const CHILD_FINAL_TEXT = "child: echo done";
15
+ export declare const FALLBACK_TEXT = "ack";
16
+ export type Step = "bg-call" | "fg-call" | "agent-call" | "agent-final" | "child-echo" | "child-final" | "fallback";
17
+ export interface GenericBlock {
18
+ type?: unknown;
19
+ tool_use_id?: unknown;
20
+ text?: unknown;
21
+ [k: string]: unknown;
22
+ }
23
+ export interface GenericMessage {
24
+ role?: unknown;
25
+ content?: unknown;
26
+ [k: string]: unknown;
27
+ }
28
+ /** True iff ANY message carries a `tool_result` content block whose `tool_use_id` matches. Never a
29
+ * text/string search -- both wire shapes (Anthropic JSON, Winter's ContentBlock[]) structurally
30
+ * agree on this field, so walking it is exact where string-matching a stringified body is not. */
31
+ export declare function hasToolResultFor(messages: readonly GenericMessage[], id: string): boolean;
32
+ /** True iff the FIRST `role: "user"` message's text contains `marker` -- the child's own
33
+ * conversation is discriminated by its first turn carrying the Agent call's `prompt` verbatim (on
34
+ * the official side it arrives wrapped in the runtime's own injected system-reminders, so this is
35
+ * a substring check on the first user turn only, never a full-conversation scan -- provider/mock.ts's
36
+ * own "subagent" case uses the identical discipline for the identical reason). */
37
+ export declare function firstUserTextIncludes(messages: readonly GenericMessage[], marker: string): boolean;
38
+ /**
39
+ * The shared decision, one call per generate()/loopback request. Mirrors exactly the sequence
40
+ * validated against the real pinned binary (see the task report): the CHILD's own conversation is
41
+ * checked first (its first user turn carries the child prompt verbatim, regardless of how many
42
+ * parent turns preceded it), then the PARENT's conversation is walked backward through its own
43
+ * scripted tool_use ids -- agent3's result means the whole script is done; fg2's result means it is
44
+ * time to spawn the agent; bg1's result means it is time for the foreground sleep; a single-message
45
+ * history means this is the very first turn. Anything else (e.g. the official runtime relaying a
46
+ * background-task notification to the model as an extra turn) falls through to a plain short reply.
47
+ */
48
+ export declare function decideStep(messages: readonly GenericMessage[]): Step;
49
+ export declare const TASK_FRAME_SUBTYPES: Set<string>;
50
+ export type RawFrame = Record<string, unknown>;
51
+ /** The STRUCTURE assertion's own per-frame reduction: {subtype, task, keys, status, task_type,
52
+ * is_backgrounded, patchKeys/patchStatus, last_tool_name, subagent_type} -- `keys` (the frame's OWN
53
+ * sorted top-level key list, post uuid/session_id drop) is what catches a field Winter fails to
54
+ * emit at all, without this file having to hand-enumerate every field per frame kind. */
55
+ export interface ReducedFrame {
56
+ subtype: string;
57
+ task: unknown;
58
+ keys: string[];
59
+ status?: unknown;
60
+ task_type?: unknown;
61
+ is_backgrounded?: unknown;
62
+ patchKeys?: string[];
63
+ patchStatus?: unknown;
64
+ last_tool_name?: unknown;
65
+ subagent_type?: unknown;
66
+ }
67
+ export interface BgSnapshotEntry {
68
+ task: unknown;
69
+ task_type: unknown;
70
+ ambient?: unknown;
71
+ }
72
+ export interface TextEntry {
73
+ subtype: string;
74
+ task: string;
75
+ field: "description" | "summary" | "prompt";
76
+ value: string;
77
+ }
78
+ export interface NormalizedProjection {
79
+ /** Every raw frame kept (system/task_*), normalized, in original emission order -- for the "print
80
+ * the full normalized sequence" requirement. */
81
+ normalized: RawFrame[];
82
+ structure: ReducedFrame[];
83
+ bgSnapshots: BgSnapshotEntry[][];
84
+ textEntries: TextEntry[];
85
+ }
86
+ /** Filters to the five task-frame subtypes, normalizes (drop uuid/session_id, relabel task ids,
87
+ * scrub numeric usage/patch.end_time/patch.total_paused_ms, scrub a non-empty output_file), and
88
+ * projects into the STRUCTURE sequence, the distinct background_tasks_changed snapshots, and the
89
+ * TEXT entries (description/summary/prompt, kept verbatim, printed but never asserted). */
90
+ export declare function normalizeAndProject(rawFrames: readonly RawFrame[]): NormalizedProjection;
91
+ export declare function formatTextEntries(label: string, entries: readonly TextEntry[]): string;
92
+ /** A key-aligned side-by-side text diff over TextEntry lists, keyed by task+subtype+field (both
93
+ * sides' task labels are independently minted by normalizeAndProject, but the SAME script produces
94
+ * the SAME task ORDER on both sides -- see the task report -- so same-position labels line up). */
95
+ export declare function diffTextEntries(officialEntries: readonly TextEntry[], winterEntries: readonly TextEntry[]): string;
@@ -47,7 +47,13 @@
47
47
  ],
48
48
  "output_style": "default",
49
49
  "skills": [],
50
- "plugins": []
50
+ "plugins": [],
51
+ "agents": [
52
+ "claude",
53
+ "Explore",
54
+ "general-purpose",
55
+ "Plan"
56
+ ]
51
57
  }
52
58
  },
53
59
  {
@@ -60,7 +66,7 @@
60
66
  "content": [
61
67
  {
62
68
  "type": "text",
63
- "text": "echo: <system-reminder>\nAuto-memory (injected by the runtime, not typed by the user):\nAuto-memory for this project lives at /winter-home/projects/-winter-fixture/memory. The directory is created on demand and there are no memory tools: read and write it with the ordinary file tools, exactly like any other directory.\n\n/winter-home/projects/-winter-fixture/memory/MEMORY.md is the INDEX, and only its first 200 lines / 25 KB are loaded into a session. Keep every index entry to one line; the substance belongs in the topic file it points at, which you can read on demand when it turns out to matter.\n\nTo record something worth having in a LATER session -- a standing preference, a correction you were given, a durable constraint of this project -- write it to its own `<slug>.md` in that directory and add a one-line pointer to the index: `- [<slug>](<slug>.md) — <one-line summary>`.\n\nRevise a fact by rewriting its file and its index line, never by adding a near-duplicate under a new name; remove both once it stops being true. An index full of stale near-duplicates is worse than an empty one, because it costs the same and misleads.\n\nDo not record what the repository already records. Code, configuration, documentation and WINTER.md are durable on their own; memory is for what is true about this project or this user and lives nowhere in the tree.\n</system-reminder>\n\nhi"
69
+ "text": "echo: <system-reminder>\nAvailable agent types for the Agent tool:\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \"src/components/**/*.tsx\"), grep for symbols or keywords (eg. \"API endpoints\"), or answer \"where is X defined / which files reference Y.\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \"quick\" for a single targeted lookup, \"medium\" for moderate exploration, or \"very thorough\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\n</system-reminder>\n\n<system-reminder>\nAs you answer the user's questions, you can use the following context:\n# currentDate\nToday's date is <FIXTURE-DATE>.\n\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\n</system-reminder>\n\n\nhi"
64
70
  }
65
71
  ]
66
72
  }
@@ -74,7 +80,7 @@
74
80
  "type": "result",
75
81
  "subtype": "success",
76
82
  "is_error": false,
77
- "result": "echo: <system-reminder>\nAuto-memory (injected by the runtime, not typed by the user):\nAuto-memory for this project lives at /winter-home/projects/-winter-fixture/memory. The directory is created on demand and there are no memory tools: read and write it with the ordinary file tools, exactly like any other directory.\n\n/winter-home/projects/-winter-fixture/memory/MEMORY.md is the INDEX, and only its first 200 lines / 25 KB are loaded into a session. Keep every index entry to one line; the substance belongs in the topic file it points at, which you can read on demand when it turns out to matter.\n\nTo record something worth having in a LATER session -- a standing preference, a correction you were given, a durable constraint of this project -- write it to its own `<slug>.md` in that directory and add a one-line pointer to the index: `- [<slug>](<slug>.md) — <one-line summary>`.\n\nRevise a fact by rewriting its file and its index line, never by adding a near-duplicate under a new name; remove both once it stops being true. An index full of stale near-duplicates is worse than an empty one, because it costs the same and misleads.\n\nDo not record what the repository already records. Code, configuration, documentation and WINTER.md are durable on their own; memory is for what is true about this project or this user and lives nowhere in the tree.\n</system-reminder>\n\nhi",
83
+ "result": "echo: <system-reminder>\nAvailable agent types for the Agent tool:\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \"src/components/**/*.tsx\"), grep for symbols or keywords (eg. \"API endpoints\"), or answer \"where is X defined / which files reference Y.\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \"quick\" for a single targeted lookup, \"medium\" for moderate exploration, or \"very thorough\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\n</system-reminder>\n\n<system-reminder>\nAs you answer the user's questions, you can use the following context:\n# currentDate\nToday's date is <FIXTURE-DATE>.\n\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\n</system-reminder>\n\n\nhi",
78
84
  "permission_denials": []
79
85
  }
80
86
  }
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -80,8 +86,10 @@
80
86
  "type": "system",
81
87
  "subtype": "task_started",
82
88
  "task_id": "TASKID",
89
+ "tool_use_id": "bash-bg-call-1",
83
90
  "description": "sleep 10",
84
- "is_backgrounded": true
91
+ "is_backgrounded": true,
92
+ "task_type": "local_bash"
85
93
  }
86
94
  },
87
95
  {
@@ -94,7 +102,7 @@
94
102
  "tasks": [
95
103
  {
96
104
  "task_id": "TASKID",
97
- "task_type": "bash",
105
+ "task_type": "local_bash",
98
106
  "description": "sleep 10"
99
107
  }
100
108
  ]
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -93,7 +93,13 @@
93
93
  ],
94
94
  "output_style": "default",
95
95
  "skills": [],
96
- "plugins": []
96
+ "plugins": [],
97
+ "agents": [
98
+ "claude",
99
+ "Explore",
100
+ "general-purpose",
101
+ "Plan"
102
+ ]
97
103
  }
98
104
  },
99
105
  {
@@ -60,6 +60,12 @@
60
60
  "name": "t8mcpsdk",
61
61
  "status": "connected"
62
62
  }
63
+ ],
64
+ "agents": [
65
+ "claude",
66
+ "Explore",
67
+ "general-purpose",
68
+ "Plan"
63
69
  ]
64
70
  }
65
71
  },
@@ -93,7 +93,13 @@
93
93
  ],
94
94
  "output_style": "default",
95
95
  "skills": [],
96
- "plugins": []
96
+ "plugins": [],
97
+ "agents": [
98
+ "claude",
99
+ "Explore",
100
+ "general-purpose",
101
+ "Plan"
102
+ ]
97
103
  }
98
104
  },
99
105
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -48,7 +48,13 @@
48
48
  ],
49
49
  "output_style": "default",
50
50
  "skills": [],
51
- "plugins": []
51
+ "plugins": [],
52
+ "agents": [
53
+ "claude",
54
+ "Explore",
55
+ "general-purpose",
56
+ "Plan"
57
+ ]
52
58
  }
53
59
  },
54
60
  {
@@ -61,7 +67,7 @@
61
67
  "content": [
62
68
  {
63
69
  "type": "text",
64
- "text": "echo: <system-reminder>\nAuto-memory (injected by the runtime, not typed by the user):\nAuto-memory for this project lives at /winter-home/projects/-winter-fixture/memory. The directory is created on demand and there are no memory tools: read and write it with the ordinary file tools, exactly like any other directory.\n\n/winter-home/projects/-winter-fixture/memory/MEMORY.md is the INDEX, and only its first 200 lines / 25 KB are loaded into a session. Keep every index entry to one line; the substance belongs in the topic file it points at, which you can read on demand when it turns out to matter.\n\nTo record something worth having in a LATER session -- a standing preference, a correction you were given, a durable constraint of this project -- write it to its own `<slug>.md` in that directory and add a one-line pointer to the index: `- [<slug>](<slug>.md) — <one-line summary>`.\n\nRevise a fact by rewriting its file and its index line, never by adding a near-duplicate under a new name; remove both once it stops being true. An index full of stale near-duplicates is worse than an empty one, because it costs the same and misleads.\n\nDo not record what the repository already records. Code, configuration, documentation and WINTER.md are durable on their own; memory is for what is true about this project or this user and lives nowhere in the tree.\n</system-reminder>\n\nfirst"
70
+ "text": "echo: <system-reminder>\nAvailable agent types for the Agent tool:\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \"src/components/**/*.tsx\"), grep for symbols or keywords (eg. \"API endpoints\"), or answer \"where is X defined / which files reference Y.\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \"quick\" for a single targeted lookup, \"medium\" for moderate exploration, or \"very thorough\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\n</system-reminder>\n\n<system-reminder>\nAs you answer the user's questions, you can use the following context:\n# currentDate\nToday's date is <FIXTURE-DATE>.\n\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\n</system-reminder>\n\n\nfirst"
65
71
  }
66
72
  ]
67
73
  }
@@ -75,7 +81,7 @@
75
81
  "type": "result",
76
82
  "subtype": "success",
77
83
  "is_error": false,
78
- "result": "echo: <system-reminder>\nAuto-memory (injected by the runtime, not typed by the user):\nAuto-memory for this project lives at /winter-home/projects/-winter-fixture/memory. The directory is created on demand and there are no memory tools: read and write it with the ordinary file tools, exactly like any other directory.\n\n/winter-home/projects/-winter-fixture/memory/MEMORY.md is the INDEX, and only its first 200 lines / 25 KB are loaded into a session. Keep every index entry to one line; the substance belongs in the topic file it points at, which you can read on demand when it turns out to matter.\n\nTo record something worth having in a LATER session -- a standing preference, a correction you were given, a durable constraint of this project -- write it to its own `<slug>.md` in that directory and add a one-line pointer to the index: `- [<slug>](<slug>.md) — <one-line summary>`.\n\nRevise a fact by rewriting its file and its index line, never by adding a near-duplicate under a new name; remove both once it stops being true. An index full of stale near-duplicates is worse than an empty one, because it costs the same and misleads.\n\nDo not record what the repository already records. Code, configuration, documentation and WINTER.md are durable on their own; memory is for what is true about this project or this user and lives nowhere in the tree.\n</system-reminder>\n\nfirst",
84
+ "result": "echo: <system-reminder>\nAvailable agent types for the Agent tool:\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \"src/components/**/*.tsx\"), grep for symbols or keywords (eg. \"API endpoints\"), or answer \"where is X defined / which files reference Y.\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \"quick\" for a single targeted lookup, \"medium\" for moderate exploration, or \"very thorough\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\n</system-reminder>\n\n<system-reminder>\nAs you answer the user's questions, you can use the following context:\n# currentDate\nToday's date is <FIXTURE-DATE>.\n\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\n</system-reminder>\n\n\nfirst",
79
85
  "permission_denials": []
80
86
  }
81
87
  },
@@ -89,7 +95,7 @@
89
95
  "content": [
90
96
  {
91
97
  "type": "text",
92
- "text": "echo: <system-reminder>\nAuto-memory (injected by the runtime, not typed by the user):\nAuto-memory for this project lives at /winter-home/projects/-winter-fixture/memory. The directory is created on demand and there are no memory tools: read and write it with the ordinary file tools, exactly like any other directory.\n\n/winter-home/projects/-winter-fixture/memory/MEMORY.md is the INDEX, and only its first 200 lines / 25 KB are loaded into a session. Keep every index entry to one line; the substance belongs in the topic file it points at, which you can read on demand when it turns out to matter.\n\nTo record something worth having in a LATER session -- a standing preference, a correction you were given, a durable constraint of this project -- write it to its own `<slug>.md` in that directory and add a one-line pointer to the index: `- [<slug>](<slug>.md) — <one-line summary>`.\n\nRevise a fact by rewriting its file and its index line, never by adding a near-duplicate under a new name; remove both once it stops being true. An index full of stale near-duplicates is worse than an empty one, because it costs the same and misleads.\n\nDo not record what the repository already records. Code, configuration, documentation and WINTER.md are durable on their own; memory is for what is true about this project or this user and lives nowhere in the tree.\n</system-reminder>\n\nsecond"
98
+ "text": "echo: second"
93
99
  }
94
100
  ]
95
101
  }
@@ -103,7 +109,7 @@
103
109
  "type": "result",
104
110
  "subtype": "success",
105
111
  "is_error": false,
106
- "result": "echo: <system-reminder>\nAuto-memory (injected by the runtime, not typed by the user):\nAuto-memory for this project lives at /winter-home/projects/-winter-fixture/memory. The directory is created on demand and there are no memory tools: read and write it with the ordinary file tools, exactly like any other directory.\n\n/winter-home/projects/-winter-fixture/memory/MEMORY.md is the INDEX, and only its first 200 lines / 25 KB are loaded into a session. Keep every index entry to one line; the substance belongs in the topic file it points at, which you can read on demand when it turns out to matter.\n\nTo record something worth having in a LATER session -- a standing preference, a correction you were given, a durable constraint of this project -- write it to its own `<slug>.md` in that directory and add a one-line pointer to the index: `- [<slug>](<slug>.md) — <one-line summary>`.\n\nRevise a fact by rewriting its file and its index line, never by adding a near-duplicate under a new name; remove both once it stops being true. An index full of stale near-duplicates is worse than an empty one, because it costs the same and misleads.\n\nDo not record what the repository already records. Code, configuration, documentation and WINTER.md are durable on their own; memory is for what is true about this project or this user and lives nowhere in the tree.\n</system-reminder>\n\nsecond",
112
+ "result": "echo: second",
107
113
  "permission_denials": []
108
114
  }
109
115
  }
@@ -58,7 +58,13 @@
58
58
  ],
59
59
  "output_style": "default",
60
60
  "skills": [],
61
- "plugins": []
61
+ "plugins": [],
62
+ "agents": [
63
+ "claude",
64
+ "Explore",
65
+ "general-purpose",
66
+ "Plan"
67
+ ]
62
68
  }
63
69
  },
64
70
  {
@@ -96,7 +102,8 @@
96
102
  {
97
103
  "type": "tool_result",
98
104
  "tool_use_id": "toolu_p6",
99
- "content": "Error: path not found: /winter-fixture"
105
+ "content": "Error: path not found: /winter-fixture",
106
+ "is_error": true
100
107
  }
101
108
  ]
102
109
  }
@@ -57,7 +57,13 @@
57
57
  ],
58
58
  "output_style": "default",
59
59
  "skills": [],
60
- "plugins": []
60
+ "plugins": [],
61
+ "agents": [
62
+ "claude",
63
+ "Explore",
64
+ "general-purpose",
65
+ "Plan"
66
+ ]
61
67
  }
62
68
  },
63
69
  {
@@ -95,7 +101,8 @@
95
101
  {
96
102
  "type": "tool_result",
97
103
  "tool_use_id": "google-call-SCRUBBED",
98
- "content": "Error: path not found: /winter-fixture"
104
+ "content": "Error: path not found: /winter-fixture",
105
+ "is_error": true
99
106
  }
100
107
  ]
101
108
  }
@@ -58,7 +58,13 @@
58
58
  ],
59
59
  "output_style": "default",
60
60
  "skills": [],
61
- "plugins": []
61
+ "plugins": [],
62
+ "agents": [
63
+ "claude",
64
+ "Explore",
65
+ "general-purpose",
66
+ "Plan"
67
+ ]
62
68
  }
63
69
  },
64
70
  {
@@ -96,7 +102,8 @@
96
102
  {
97
103
  "type": "tool_result",
98
104
  "tool_use_id": "call_p6",
99
- "content": "Error: path not found: /winter-fixture"
105
+ "content": "Error: path not found: /winter-fixture",
106
+ "is_error": true
100
107
  }
101
108
  ]
102
109
  }