@yanlinglabs/winter-conformance 0.0.17 → 0.0.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/goldens/advertised-set-round.trace.json +33 -1
  2. package/goldens/background-task-round.trace.json +33 -1
  3. package/goldens/bash-background-round.trace.json +33 -1
  4. package/goldens/canusetool-approved-round.trace.json +33 -1
  5. package/goldens/compaction-auto-round.trace.json +165 -5
  6. package/goldens/compaction-manual-round.trace.json +186 -5
  7. package/goldens/denied-tool-round.trace.json +33 -1
  8. package/goldens/hook-denied-round.trace.json +33 -1
  9. package/goldens/hooked-tool-round.trace.json +33 -1
  10. package/goldens/interrupt.trace.json +21 -0
  11. package/goldens/mcp-tool-round.trace.json +33 -1
  12. package/goldens/mode-switch-mid-session.trace.json +66 -2
  13. package/goldens/multi-turn.trace.json +66 -2
  14. package/goldens/p6-anthropic-fake.trace.json +25 -4
  15. package/goldens/p6-gemini-fake.trace.json +35 -1
  16. package/goldens/p6-openai-chat-fake.trace.json +36 -1
  17. package/goldens/p6-openai-responses-fake.trace.json +26 -4
  18. package/goldens/p6-resolution-failure.trace.json +21 -0
  19. package/goldens/plain-query.trace.json +33 -1
  20. package/goldens/resume.trace.json +66 -2
  21. package/goldens/sendmessage-child-round.trace.json +33 -1
  22. package/goldens/skill-invocation-round.trace.json +34 -2
  23. package/goldens/structured-exhaustion-round.trace.json +33 -1
  24. package/goldens/structured-output-round.trace.json +33 -1
  25. package/goldens/subagent-permission-round.trace.json +33 -1
  26. package/goldens/subagent-spawn-round.trace.json +33 -1
  27. package/goldens/tool-round.trace.json +33 -1
  28. package/goldens/toolsearch-select-round.trace.json +33 -1
  29. package/package.json +1 -1
@@ -84,7 +84,39 @@
84
84
  "subtype": "success",
85
85
  "is_error": false,
86
86
  "result": "[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"<system-reminder>\\nAvailable agent types for the Agent tool:\\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \\\"src/components/**/*.tsx\\\"), grep for symbols or keywords (eg. \\\"API endpoints\\\"), or answer \\\"where is X defined / which files reference Y.\\\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \\\"quick\\\" for a single targeted lookup, \\\"medium\\\" for moderate exploration, or \\\"very thorough\\\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n\\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\\n</system-reminder>\\n\"},{\"type\":\"text\",\"text\":\"<system-reminder>\\nAs you answer the user's questions, you can use the following context:\\n# currentDate\\nToday's date is <FIXTURE-DATE>.\\n\\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\\n</system-reminder>\\n\\n\"},{\"type\":\"text\",\"text\":\"first\"}]}]",
87
- "permission_denials": []
87
+ "usage": {
88
+ "output_tokens_details": {
89
+ "thinking_tokens": 0
90
+ },
91
+ "input_tokens": "<input_tokens>",
92
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
93
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
94
+ "output_tokens": "<output_tokens>",
95
+ "server_tool_use": {
96
+ "web_search_requests": 0,
97
+ "web_fetch_requests": 0
98
+ },
99
+ "service_tier": "standard",
100
+ "cache_creation": {
101
+ "ephemeral_1h_input_tokens": 0,
102
+ "ephemeral_5m_input_tokens": 0
103
+ },
104
+ "inference_geo": "",
105
+ "iterations": [],
106
+ "speed": "standard"
107
+ },
108
+ "permission_denials": [],
109
+ "modelUsage": {
110
+ "winter-test/echo": {
111
+ "inputTokens": "<inputTokens>",
112
+ "outputTokens": "<outputTokens>",
113
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
114
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
115
+ "webSearchRequests": 0,
116
+ "costUSD": 0,
117
+ "canonicalModel": "winter-test/echo"
118
+ }
119
+ }
88
120
  }
89
121
  },
90
122
  {
@@ -172,7 +204,39 @@
172
204
  "subtype": "success",
173
205
  "is_error": false,
174
206
  "result": "[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"<system-reminder>\\nAvailable agent types for the Agent tool:\\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \\\"src/components/**/*.tsx\\\"), grep for symbols or keywords (eg. \\\"API endpoints\\\"), or answer \\\"where is X defined / which files reference Y.\\\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \\\"quick\\\" for a single targeted lookup, \\\"medium\\\" for moderate exploration, or \\\"very thorough\\\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n\\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\\n</system-reminder>\\n\"},{\"type\":\"text\",\"text\":\"<system-reminder>\\nAs you answer the user's questions, you can use the following context:\\n# currentDate\\nToday's date is <FIXTURE-DATE>.\\n\\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\\n</system-reminder>\\n\\n\"},{\"type\":\"text\",\"text\":\"first\"}]},{\"role\":\"assistant\",\"content\":\"[{\\\"role\\\":\\\"user\\\",\\\"content\\\":[{\\\"type\\\":\\\"text\\\",\\\"text\\\":\\\"<system-reminder>\\\\nAvailable agent types for the Agent tool:\\\\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\\\\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \\\\\\\"src/components/**/*.tsx\\\\\\\"), grep for symbols or keywords (eg. \\\\\\\"API endpoints\\\\\\\"), or answer \\\\\\\"where is X defined / which files reference Y.\\\\\\\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \\\\\\\"quick\\\\\\\" for a single targeted lookup, \\\\\\\"medium\\\\\\\" for moderate exploration, or \\\\\\\"very thorough\\\\\\\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\\\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\\\\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\\\n\\\\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\\\\n</system-reminder>\\\\n\\\"},{\\\"type\\\":\\\"text\\\",\\\"text\\\":\\\"<system-reminder>\\\\nAs you answer the user's questions, you can use the following context:\\\\n# currentDate\\\\nToday's date is <FIXTURE-DATE>.\\\\n\\\\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\\\\n</system-reminder>\\\\n\\\\n\\\"},{\\\"type\\\":\\\"text\\\",\\\"text\\\":\\\"first\\\"}]}]\"},{\"role\":\"user\",\"content\":\"second\"}]",
175
- "permission_denials": []
207
+ "usage": {
208
+ "output_tokens_details": {
209
+ "thinking_tokens": 0
210
+ },
211
+ "input_tokens": "<input_tokens>",
212
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
213
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
214
+ "output_tokens": "<output_tokens>",
215
+ "server_tool_use": {
216
+ "web_search_requests": 0,
217
+ "web_fetch_requests": 0
218
+ },
219
+ "service_tier": "standard",
220
+ "cache_creation": {
221
+ "ephemeral_1h_input_tokens": 0,
222
+ "ephemeral_5m_input_tokens": 0
223
+ },
224
+ "inference_geo": "",
225
+ "iterations": [],
226
+ "speed": "standard"
227
+ },
228
+ "permission_denials": [],
229
+ "modelUsage": {
230
+ "winter-test/echo": {
231
+ "inputTokens": "<inputTokens>",
232
+ "outputTokens": "<outputTokens>",
233
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
234
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
235
+ "webSearchRequests": 0,
236
+ "costUSD": 0,
237
+ "canonicalModel": "winter-test/echo"
238
+ }
239
+ }
176
240
  }
177
241
  }
178
242
  ]
@@ -210,7 +210,39 @@
210
210
  "subtype": "success",
211
211
  "is_error": false,
212
212
  "result": "messaging done",
213
- "permission_denials": []
213
+ "usage": {
214
+ "output_tokens_details": {
215
+ "thinking_tokens": 0
216
+ },
217
+ "input_tokens": "<input_tokens>",
218
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
219
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
220
+ "output_tokens": "<output_tokens>",
221
+ "server_tool_use": {
222
+ "web_search_requests": 0,
223
+ "web_fetch_requests": 0
224
+ },
225
+ "service_tier": "standard",
226
+ "cache_creation": {
227
+ "ephemeral_1h_input_tokens": 0,
228
+ "ephemeral_5m_input_tokens": 0
229
+ },
230
+ "inference_geo": "",
231
+ "iterations": [],
232
+ "speed": "standard"
233
+ },
234
+ "permission_denials": [],
235
+ "modelUsage": {
236
+ "winter-test/echo": {
237
+ "inputTokens": "<inputTokens>",
238
+ "outputTokens": "<outputTokens>",
239
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
240
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
241
+ "webSearchRequests": 0,
242
+ "costUSD": 0,
243
+ "canonicalModel": "winter-test/echo"
244
+ }
245
+ }
214
246
  }
215
247
  }
216
248
  ]
@@ -94,7 +94,7 @@
94
94
  {
95
95
  "type": "tool_result",
96
96
  "tool_use_id": "p5-skill-1",
97
- "content": "P5 SKILL BODY MARKER\n"
97
+ "content": "Base directory for this skill: /winter-home/skills/p5probe\n\nP5 SKILL BODY MARKER\n"
98
98
  }
99
99
  ]
100
100
  }
@@ -125,7 +125,39 @@
125
125
  "subtype": "success",
126
126
  "is_error": false,
127
127
  "result": "skill done",
128
- "permission_denials": []
128
+ "usage": {
129
+ "output_tokens_details": {
130
+ "thinking_tokens": 0
131
+ },
132
+ "input_tokens": "<input_tokens>",
133
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
134
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
135
+ "output_tokens": "<output_tokens>",
136
+ "server_tool_use": {
137
+ "web_search_requests": 0,
138
+ "web_fetch_requests": 0
139
+ },
140
+ "service_tier": "standard",
141
+ "cache_creation": {
142
+ "ephemeral_1h_input_tokens": 0,
143
+ "ephemeral_5m_input_tokens": 0
144
+ },
145
+ "inference_geo": "",
146
+ "iterations": [],
147
+ "speed": "standard"
148
+ },
149
+ "permission_denials": [],
150
+ "modelUsage": {
151
+ "winter-test/echo": {
152
+ "inputTokens": "<inputTokens>",
153
+ "outputTokens": "<outputTokens>",
154
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
155
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
156
+ "webSearchRequests": 0,
157
+ "costUSD": 0,
158
+ "canonicalModel": "winter-test/echo"
159
+ }
160
+ }
129
161
  }
130
162
  }
131
163
  ]
@@ -184,7 +184,39 @@
184
184
  "is_error": true,
185
185
  "result": "Failed to provide valid structured output after 3 attempts",
186
186
  "terminal_reason": "structured_output_retry_exhausted",
187
- "permission_denials": []
187
+ "usage": {
188
+ "output_tokens_details": {
189
+ "thinking_tokens": 0
190
+ },
191
+ "input_tokens": "<input_tokens>",
192
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
193
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
194
+ "output_tokens": "<output_tokens>",
195
+ "server_tool_use": {
196
+ "web_search_requests": 0,
197
+ "web_fetch_requests": 0
198
+ },
199
+ "service_tier": "standard",
200
+ "cache_creation": {
201
+ "ephemeral_1h_input_tokens": 0,
202
+ "ephemeral_5m_input_tokens": 0
203
+ },
204
+ "inference_geo": "",
205
+ "iterations": [],
206
+ "speed": "standard"
207
+ },
208
+ "permission_denials": [],
209
+ "modelUsage": {
210
+ "winter-test/echo": {
211
+ "inputTokens": "<inputTokens>",
212
+ "outputTokens": "<outputTokens>",
213
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
214
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
215
+ "webSearchRequests": 0,
216
+ "costUSD": 0,
217
+ "canonicalModel": "winter-test/echo"
218
+ }
219
+ }
188
220
  }
189
221
  }
190
222
  ]
@@ -108,7 +108,39 @@
108
108
  "structured_output": {
109
109
  "answer": 42
110
110
  },
111
- "permission_denials": []
111
+ "usage": {
112
+ "output_tokens_details": {
113
+ "thinking_tokens": 0
114
+ },
115
+ "input_tokens": "<input_tokens>",
116
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
117
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
118
+ "output_tokens": "<output_tokens>",
119
+ "server_tool_use": {
120
+ "web_search_requests": 0,
121
+ "web_fetch_requests": 0
122
+ },
123
+ "service_tier": "standard",
124
+ "cache_creation": {
125
+ "ephemeral_1h_input_tokens": 0,
126
+ "ephemeral_5m_input_tokens": 0
127
+ },
128
+ "inference_geo": "",
129
+ "iterations": [],
130
+ "speed": "standard"
131
+ },
132
+ "permission_denials": [],
133
+ "modelUsage": {
134
+ "winter-test/echo": {
135
+ "inputTokens": "<inputTokens>",
136
+ "outputTokens": "<outputTokens>",
137
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
138
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
139
+ "webSearchRequests": 0,
140
+ "costUSD": 0,
141
+ "canonicalModel": "winter-test/echo"
142
+ }
143
+ }
112
144
  }
113
145
  }
114
146
  ]
@@ -244,7 +244,39 @@
244
244
  "subtype": "success",
245
245
  "is_error": false,
246
246
  "result": "parent finished",
247
- "permission_denials": []
247
+ "usage": {
248
+ "output_tokens_details": {
249
+ "thinking_tokens": 0
250
+ },
251
+ "input_tokens": "<input_tokens>",
252
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
253
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
254
+ "output_tokens": "<output_tokens>",
255
+ "server_tool_use": {
256
+ "web_search_requests": 0,
257
+ "web_fetch_requests": 0
258
+ },
259
+ "service_tier": "standard",
260
+ "cache_creation": {
261
+ "ephemeral_1h_input_tokens": 0,
262
+ "ephemeral_5m_input_tokens": 0
263
+ },
264
+ "inference_geo": "",
265
+ "iterations": [],
266
+ "speed": "standard"
267
+ },
268
+ "permission_denials": [],
269
+ "modelUsage": {
270
+ "winter-test/echo": {
271
+ "inputTokens": "<inputTokens>",
272
+ "outputTokens": "<outputTokens>",
273
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
274
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
275
+ "webSearchRequests": 0,
276
+ "costUSD": 0,
277
+ "canonicalModel": "winter-test/echo"
278
+ }
279
+ }
248
280
  }
249
281
  }
250
282
  ]
@@ -172,7 +172,39 @@
172
172
  "subtype": "success",
173
173
  "is_error": false,
174
174
  "result": "parent finished",
175
- "permission_denials": []
175
+ "usage": {
176
+ "output_tokens_details": {
177
+ "thinking_tokens": 0
178
+ },
179
+ "input_tokens": "<input_tokens>",
180
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
181
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
182
+ "output_tokens": "<output_tokens>",
183
+ "server_tool_use": {
184
+ "web_search_requests": 0,
185
+ "web_fetch_requests": 0
186
+ },
187
+ "service_tier": "standard",
188
+ "cache_creation": {
189
+ "ephemeral_1h_input_tokens": 0,
190
+ "ephemeral_5m_input_tokens": 0
191
+ },
192
+ "inference_geo": "",
193
+ "iterations": [],
194
+ "speed": "standard"
195
+ },
196
+ "permission_denials": [],
197
+ "modelUsage": {
198
+ "winter-test/echo": {
199
+ "inputTokens": "<inputTokens>",
200
+ "outputTokens": "<outputTokens>",
201
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
202
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
203
+ "webSearchRequests": 0,
204
+ "costUSD": 0,
205
+ "canonicalModel": "winter-test/echo"
206
+ }
207
+ }
176
208
  }
177
209
  }
178
210
  ]
@@ -121,7 +121,39 @@
121
121
  "subtype": "success",
122
122
  "is_error": false,
123
123
  "result": "tool round done",
124
- "permission_denials": []
124
+ "usage": {
125
+ "output_tokens_details": {
126
+ "thinking_tokens": 0
127
+ },
128
+ "input_tokens": "<input_tokens>",
129
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
130
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
131
+ "output_tokens": "<output_tokens>",
132
+ "server_tool_use": {
133
+ "web_search_requests": 0,
134
+ "web_fetch_requests": 0
135
+ },
136
+ "service_tier": "standard",
137
+ "cache_creation": {
138
+ "ephemeral_1h_input_tokens": 0,
139
+ "ephemeral_5m_input_tokens": 0
140
+ },
141
+ "inference_geo": "",
142
+ "iterations": [],
143
+ "speed": "standard"
144
+ },
145
+ "permission_denials": [],
146
+ "modelUsage": {
147
+ "winter-test/echo": {
148
+ "inputTokens": "<inputTokens>",
149
+ "outputTokens": "<outputTokens>",
150
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
151
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
152
+ "webSearchRequests": 0,
153
+ "costUSD": 0,
154
+ "canonicalModel": "winter-test/echo"
155
+ }
156
+ }
125
157
  }
126
158
  }
127
159
  ]
@@ -187,7 +187,39 @@
187
187
  "subtype": "success",
188
188
  "is_error": false,
189
189
  "result": "tool search done",
190
- "permission_denials": []
190
+ "usage": {
191
+ "output_tokens_details": {
192
+ "thinking_tokens": 0
193
+ },
194
+ "input_tokens": "<input_tokens>",
195
+ "cache_creation_input_tokens": "<cache_creation_input_tokens>",
196
+ "cache_read_input_tokens": "<cache_read_input_tokens>",
197
+ "output_tokens": "<output_tokens>",
198
+ "server_tool_use": {
199
+ "web_search_requests": 0,
200
+ "web_fetch_requests": 0
201
+ },
202
+ "service_tier": "standard",
203
+ "cache_creation": {
204
+ "ephemeral_1h_input_tokens": 0,
205
+ "ephemeral_5m_input_tokens": 0
206
+ },
207
+ "inference_geo": "",
208
+ "iterations": [],
209
+ "speed": "standard"
210
+ },
211
+ "permission_denials": [],
212
+ "modelUsage": {
213
+ "winter-test/echo": {
214
+ "inputTokens": "<inputTokens>",
215
+ "outputTokens": "<outputTokens>",
216
+ "cacheReadInputTokens": "<cacheReadInputTokens>",
217
+ "cacheCreationInputTokens": "<cacheCreationInputTokens>",
218
+ "webSearchRequests": 0,
219
+ "costUSD": 0,
220
+ "canonicalModel": "winter-test/echo"
221
+ }
222
+ }
191
223
  }
192
224
  }
193
225
  ]
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yanlinglabs/winter-conformance",
3
- "version": "0.0.17",
3
+ "version": "0.0.22",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "engines": {