@yanlinglabs/winter-conformance 0.0.16 → 0.0.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/official/web-tools-script.d.ts +118 -0
- package/goldens/advertised-set-round.trace.json +35 -1
- package/goldens/background-task-round.trace.json +35 -1
- package/goldens/bash-background-round.trace.json +35 -1
- package/goldens/canusetool-approved-round.trace.json +35 -1
- package/goldens/compaction-auto-round.trace.json +167 -5
- package/goldens/compaction-manual-round.trace.json +188 -5
- package/goldens/denied-tool-round.trace.json +35 -1
- package/goldens/hook-denied-round.trace.json +35 -1
- package/goldens/hooked-tool-round.trace.json +35 -1
- package/goldens/interrupt.trace.json +25 -0
- package/goldens/mcp-tool-round.trace.json +35 -1
- package/goldens/messaging-facet-round.trace.json +4 -0
- package/goldens/mode-switch-mid-session.trace.json +68 -2
- package/goldens/multi-turn.trace.json +68 -2
- package/goldens/p6-anthropic-fake.trace.json +27 -4
- package/goldens/p6-gemini-fake.trace.json +36 -1
- package/goldens/p6-openai-chat-fake.trace.json +37 -1
- package/goldens/p6-openai-responses-fake.trace.json +27 -4
- package/goldens/p6-resolution-failure.trace.json +23 -0
- package/goldens/plain-query.trace.json +35 -1
- package/goldens/resume.trace.json +70 -2
- package/goldens/sendmessage-child-round.trace.json +35 -1
- package/goldens/skill-invocation-round.trace.json +36 -2
- package/goldens/structured-exhaustion-round.trace.json +35 -1
- package/goldens/structured-output-round.trace.json +35 -1
- package/goldens/subagent-permission-round.trace.json +35 -1
- package/goldens/subagent-spawn-round.trace.json +35 -1
- package/goldens/tool-round.trace.json +35 -1
- package/goldens/toolsearch-select-round.trace.json +35 -1
- package/package.json +1 -1
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
"TaskOutput",
|
|
40
40
|
"TaskStop",
|
|
41
41
|
"TaskUpdate",
|
|
42
|
+
"WebFetch",
|
|
43
|
+
"WebSearch",
|
|
42
44
|
"Workflow",
|
|
43
45
|
"Write"
|
|
44
46
|
],
|
|
@@ -82,7 +84,39 @@
|
|
|
82
84
|
"subtype": "success",
|
|
83
85
|
"is_error": false,
|
|
84
86
|
"result": "echo: <system-reminder>\nAvailable agent types for the Agent tool:\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \"src/components/**/*.tsx\"), grep for symbols or keywords (eg. \"API endpoints\"), or answer \"where is X defined / which files reference Y.\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \"quick\" for a single targeted lookup, \"medium\" for moderate exploration, or \"very thorough\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\n</system-reminder>\n\n<system-reminder>\nAs you answer the user's questions, you can use the following context:\n# currentDate\nToday's date is <FIXTURE-DATE>.\n\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\n</system-reminder>\n\n\nfirst",
|
|
85
|
-
"
|
|
87
|
+
"usage": {
|
|
88
|
+
"output_tokens_details": {
|
|
89
|
+
"thinking_tokens": 0
|
|
90
|
+
},
|
|
91
|
+
"input_tokens": "<input_tokens>",
|
|
92
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
93
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
94
|
+
"output_tokens": "<output_tokens>",
|
|
95
|
+
"server_tool_use": {
|
|
96
|
+
"web_search_requests": 0,
|
|
97
|
+
"web_fetch_requests": 0
|
|
98
|
+
},
|
|
99
|
+
"service_tier": "standard",
|
|
100
|
+
"cache_creation": {
|
|
101
|
+
"ephemeral_1h_input_tokens": 0,
|
|
102
|
+
"ephemeral_5m_input_tokens": 0
|
|
103
|
+
},
|
|
104
|
+
"inference_geo": "",
|
|
105
|
+
"iterations": [],
|
|
106
|
+
"speed": "standard"
|
|
107
|
+
},
|
|
108
|
+
"permission_denials": [],
|
|
109
|
+
"modelUsage": {
|
|
110
|
+
"winter-test/echo": {
|
|
111
|
+
"inputTokens": "<inputTokens>",
|
|
112
|
+
"outputTokens": "<outputTokens>",
|
|
113
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
114
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
115
|
+
"webSearchRequests": 0,
|
|
116
|
+
"costUSD": 0,
|
|
117
|
+
"canonicalModel": "winter-test/echo"
|
|
118
|
+
}
|
|
119
|
+
}
|
|
86
120
|
}
|
|
87
121
|
},
|
|
88
122
|
{
|
|
@@ -110,7 +144,39 @@
|
|
|
110
144
|
"subtype": "success",
|
|
111
145
|
"is_error": false,
|
|
112
146
|
"result": "echo: second",
|
|
113
|
-
"
|
|
147
|
+
"usage": {
|
|
148
|
+
"output_tokens_details": {
|
|
149
|
+
"thinking_tokens": 0
|
|
150
|
+
},
|
|
151
|
+
"input_tokens": "<input_tokens>",
|
|
152
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
153
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
154
|
+
"output_tokens": "<output_tokens>",
|
|
155
|
+
"server_tool_use": {
|
|
156
|
+
"web_search_requests": 0,
|
|
157
|
+
"web_fetch_requests": 0
|
|
158
|
+
},
|
|
159
|
+
"service_tier": "standard",
|
|
160
|
+
"cache_creation": {
|
|
161
|
+
"ephemeral_1h_input_tokens": 0,
|
|
162
|
+
"ephemeral_5m_input_tokens": 0
|
|
163
|
+
},
|
|
164
|
+
"inference_geo": "",
|
|
165
|
+
"iterations": [],
|
|
166
|
+
"speed": "standard"
|
|
167
|
+
},
|
|
168
|
+
"permission_denials": [],
|
|
169
|
+
"modelUsage": {
|
|
170
|
+
"winter-test/echo": {
|
|
171
|
+
"inputTokens": "<inputTokens>",
|
|
172
|
+
"outputTokens": "<outputTokens>",
|
|
173
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
174
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
175
|
+
"webSearchRequests": 0,
|
|
176
|
+
"costUSD": 0,
|
|
177
|
+
"canonicalModel": "winter-test/echo"
|
|
178
|
+
}
|
|
179
|
+
}
|
|
114
180
|
}
|
|
115
181
|
}
|
|
116
182
|
]
|
|
@@ -40,6 +40,8 @@
|
|
|
40
40
|
"TaskOutput",
|
|
41
41
|
"TaskStop",
|
|
42
42
|
"TaskUpdate",
|
|
43
|
+
"WebFetch",
|
|
44
|
+
"WebSearch",
|
|
43
45
|
"Workflow",
|
|
44
46
|
"Write"
|
|
45
47
|
],
|
|
@@ -134,13 +136,34 @@
|
|
|
134
136
|
"subtype": "success",
|
|
135
137
|
"is_error": false,
|
|
136
138
|
"result": "the provider scenario is done",
|
|
139
|
+
"usage": {
|
|
140
|
+
"output_tokens_details": {
|
|
141
|
+
"thinking_tokens": 0
|
|
142
|
+
},
|
|
143
|
+
"input_tokens": "<input_tokens>",
|
|
144
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
145
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
146
|
+
"output_tokens": "<output_tokens>",
|
|
147
|
+
"server_tool_use": {
|
|
148
|
+
"web_search_requests": 0,
|
|
149
|
+
"web_fetch_requests": 0
|
|
150
|
+
},
|
|
151
|
+
"service_tier": "standard",
|
|
152
|
+
"cache_creation": {
|
|
153
|
+
"ephemeral_1h_input_tokens": 0,
|
|
154
|
+
"ephemeral_5m_input_tokens": 0
|
|
155
|
+
},
|
|
156
|
+
"inference_geo": "",
|
|
157
|
+
"iterations": [],
|
|
158
|
+
"speed": "standard"
|
|
159
|
+
},
|
|
137
160
|
"permission_denials": [],
|
|
138
161
|
"modelUsage": {
|
|
139
162
|
"anthropic/claude-sonnet-5": {
|
|
140
|
-
"inputTokens":
|
|
141
|
-
"outputTokens":
|
|
142
|
-
"cacheReadInputTokens":
|
|
143
|
-
"cacheCreationInputTokens":
|
|
163
|
+
"inputTokens": "<inputTokens>",
|
|
164
|
+
"outputTokens": "<outputTokens>",
|
|
165
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
166
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
144
167
|
"webSearchRequests": 0,
|
|
145
168
|
"costUSD": 0.000098,
|
|
146
169
|
"contextWindow": 1048576,
|
|
@@ -40,6 +40,8 @@
|
|
|
40
40
|
"TaskOutput",
|
|
41
41
|
"TaskStop",
|
|
42
42
|
"TaskUpdate",
|
|
43
|
+
"WebFetch",
|
|
44
|
+
"WebSearch",
|
|
43
45
|
"Workflow",
|
|
44
46
|
"Write"
|
|
45
47
|
],
|
|
@@ -133,7 +135,40 @@
|
|
|
133
135
|
"subtype": "success",
|
|
134
136
|
"is_error": false,
|
|
135
137
|
"result": "the provider scenario is done",
|
|
136
|
-
"
|
|
138
|
+
"usage": {
|
|
139
|
+
"output_tokens_details": {
|
|
140
|
+
"thinking_tokens": 0
|
|
141
|
+
},
|
|
142
|
+
"input_tokens": "<input_tokens>",
|
|
143
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
144
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
145
|
+
"output_tokens": "<output_tokens>",
|
|
146
|
+
"server_tool_use": {
|
|
147
|
+
"web_search_requests": 0,
|
|
148
|
+
"web_fetch_requests": 0
|
|
149
|
+
},
|
|
150
|
+
"service_tier": "standard",
|
|
151
|
+
"cache_creation": {
|
|
152
|
+
"ephemeral_1h_input_tokens": 0,
|
|
153
|
+
"ephemeral_5m_input_tokens": 0
|
|
154
|
+
},
|
|
155
|
+
"inference_geo": "",
|
|
156
|
+
"iterations": [],
|
|
157
|
+
"speed": "standard"
|
|
158
|
+
},
|
|
159
|
+
"permission_denials": [],
|
|
160
|
+
"modelUsage": {
|
|
161
|
+
"google/gemini-2.5-flash": {
|
|
162
|
+
"inputTokens": "<inputTokens>",
|
|
163
|
+
"outputTokens": "<outputTokens>",
|
|
164
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
165
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
166
|
+
"webSearchRequests": 0,
|
|
167
|
+
"costUSD": 0,
|
|
168
|
+
"contextWindow": 1048576,
|
|
169
|
+
"canonicalModel": "google/gemini-2.5-flash"
|
|
170
|
+
}
|
|
171
|
+
}
|
|
137
172
|
}
|
|
138
173
|
}
|
|
139
174
|
]
|
|
@@ -40,6 +40,8 @@
|
|
|
40
40
|
"TaskOutput",
|
|
41
41
|
"TaskStop",
|
|
42
42
|
"TaskUpdate",
|
|
43
|
+
"WebFetch",
|
|
44
|
+
"WebSearch",
|
|
43
45
|
"Workflow",
|
|
44
46
|
"Write"
|
|
45
47
|
],
|
|
@@ -134,7 +136,41 @@
|
|
|
134
136
|
"subtype": "success",
|
|
135
137
|
"is_error": false,
|
|
136
138
|
"result": "the provider scenario is done",
|
|
137
|
-
"
|
|
139
|
+
"usage": {
|
|
140
|
+
"output_tokens_details": {
|
|
141
|
+
"thinking_tokens": 0
|
|
142
|
+
},
|
|
143
|
+
"input_tokens": "<input_tokens>",
|
|
144
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
145
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
146
|
+
"output_tokens": "<output_tokens>",
|
|
147
|
+
"server_tool_use": {
|
|
148
|
+
"web_search_requests": 0,
|
|
149
|
+
"web_fetch_requests": 0
|
|
150
|
+
},
|
|
151
|
+
"service_tier": "standard",
|
|
152
|
+
"cache_creation": {
|
|
153
|
+
"ephemeral_1h_input_tokens": 0,
|
|
154
|
+
"ephemeral_5m_input_tokens": 0
|
|
155
|
+
},
|
|
156
|
+
"inference_geo": "",
|
|
157
|
+
"iterations": [],
|
|
158
|
+
"speed": "standard"
|
|
159
|
+
},
|
|
160
|
+
"permission_denials": [],
|
|
161
|
+
"modelUsage": {
|
|
162
|
+
"deepseek/deepseek-v4-pro": {
|
|
163
|
+
"inputTokens": "<inputTokens>",
|
|
164
|
+
"outputTokens": "<outputTokens>",
|
|
165
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
166
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
167
|
+
"webSearchRequests": 0,
|
|
168
|
+
"costUSD": 0,
|
|
169
|
+
"contextWindow": 1000000,
|
|
170
|
+
"maxOutputTokens": 384000,
|
|
171
|
+
"canonicalModel": "deepseek/deepseek-v4-pro"
|
|
172
|
+
}
|
|
173
|
+
}
|
|
138
174
|
}
|
|
139
175
|
}
|
|
140
176
|
]
|
|
@@ -40,6 +40,8 @@
|
|
|
40
40
|
"TaskOutput",
|
|
41
41
|
"TaskStop",
|
|
42
42
|
"TaskUpdate",
|
|
43
|
+
"WebFetch",
|
|
44
|
+
"WebSearch",
|
|
43
45
|
"Workflow",
|
|
44
46
|
"Write"
|
|
45
47
|
],
|
|
@@ -133,13 +135,34 @@
|
|
|
133
135
|
"subtype": "success",
|
|
134
136
|
"is_error": false,
|
|
135
137
|
"result": "the provider scenario is done",
|
|
138
|
+
"usage": {
|
|
139
|
+
"output_tokens_details": {
|
|
140
|
+
"thinking_tokens": 0
|
|
141
|
+
},
|
|
142
|
+
"input_tokens": "<input_tokens>",
|
|
143
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
144
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
145
|
+
"output_tokens": "<output_tokens>",
|
|
146
|
+
"server_tool_use": {
|
|
147
|
+
"web_search_requests": 0,
|
|
148
|
+
"web_fetch_requests": 0
|
|
149
|
+
},
|
|
150
|
+
"service_tier": "standard",
|
|
151
|
+
"cache_creation": {
|
|
152
|
+
"ephemeral_1h_input_tokens": 0,
|
|
153
|
+
"ephemeral_5m_input_tokens": 0
|
|
154
|
+
},
|
|
155
|
+
"inference_geo": "",
|
|
156
|
+
"iterations": [],
|
|
157
|
+
"speed": "standard"
|
|
158
|
+
},
|
|
136
159
|
"permission_denials": [],
|
|
137
160
|
"modelUsage": {
|
|
138
161
|
"openai/gpt-4.1": {
|
|
139
|
-
"inputTokens":
|
|
140
|
-
"outputTokens":
|
|
141
|
-
"cacheReadInputTokens":
|
|
142
|
-
"cacheCreationInputTokens":
|
|
162
|
+
"inputTokens": "<inputTokens>",
|
|
163
|
+
"outputTokens": "<outputTokens>",
|
|
164
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
165
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
143
166
|
"webSearchRequests": 0,
|
|
144
167
|
"costUSD": 0.000092,
|
|
145
168
|
"contextWindow": 1047576,
|
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
"TaskOutput",
|
|
40
40
|
"TaskStop",
|
|
41
41
|
"TaskUpdate",
|
|
42
|
+
"WebFetch",
|
|
43
|
+
"WebSearch",
|
|
42
44
|
"Workflow",
|
|
43
45
|
"Write"
|
|
44
46
|
],
|
|
@@ -68,6 +70,27 @@
|
|
|
68
70
|
"result": "model \"definitely-not-a-model-t10\" is not in provider \"anthropic\"'s catalog (its live catalog is authoritative, so absence is definitive)",
|
|
69
71
|
"terminal_reason": "api_error",
|
|
70
72
|
"api_error_status": null,
|
|
73
|
+
"usage": {
|
|
74
|
+
"output_tokens_details": {
|
|
75
|
+
"thinking_tokens": 0
|
|
76
|
+
},
|
|
77
|
+
"input_tokens": "<input_tokens>",
|
|
78
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
79
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
80
|
+
"output_tokens": "<output_tokens>",
|
|
81
|
+
"server_tool_use": {
|
|
82
|
+
"web_search_requests": 0,
|
|
83
|
+
"web_fetch_requests": 0
|
|
84
|
+
},
|
|
85
|
+
"service_tier": "standard",
|
|
86
|
+
"cache_creation": {
|
|
87
|
+
"ephemeral_1h_input_tokens": 0,
|
|
88
|
+
"ephemeral_5m_input_tokens": 0
|
|
89
|
+
},
|
|
90
|
+
"inference_geo": "",
|
|
91
|
+
"iterations": [],
|
|
92
|
+
"speed": "standard"
|
|
93
|
+
},
|
|
71
94
|
"permission_denials": []
|
|
72
95
|
}
|
|
73
96
|
}
|
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
"TaskOutput",
|
|
40
40
|
"TaskStop",
|
|
41
41
|
"TaskUpdate",
|
|
42
|
+
"WebFetch",
|
|
43
|
+
"WebSearch",
|
|
42
44
|
"Workflow",
|
|
43
45
|
"Write"
|
|
44
46
|
],
|
|
@@ -82,7 +84,39 @@
|
|
|
82
84
|
"subtype": "success",
|
|
83
85
|
"is_error": false,
|
|
84
86
|
"result": "echo: <system-reminder>\nAvailable agent types for the Agent tool:\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \"src/components/**/*.tsx\"), grep for symbols or keywords (eg. \"API endpoints\"), or answer \"where is X defined / which files reference Y.\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \"quick\" for a single targeted lookup, \"medium\" for moderate exploration, or \"very thorough\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\n\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\n</system-reminder>\n\n<system-reminder>\nAs you answer the user's questions, you can use the following context:\n# currentDate\nToday's date is <FIXTURE-DATE>.\n\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\n</system-reminder>\n\n\nhi",
|
|
85
|
-
"
|
|
87
|
+
"usage": {
|
|
88
|
+
"output_tokens_details": {
|
|
89
|
+
"thinking_tokens": 0
|
|
90
|
+
},
|
|
91
|
+
"input_tokens": "<input_tokens>",
|
|
92
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
93
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
94
|
+
"output_tokens": "<output_tokens>",
|
|
95
|
+
"server_tool_use": {
|
|
96
|
+
"web_search_requests": 0,
|
|
97
|
+
"web_fetch_requests": 0
|
|
98
|
+
},
|
|
99
|
+
"service_tier": "standard",
|
|
100
|
+
"cache_creation": {
|
|
101
|
+
"ephemeral_1h_input_tokens": 0,
|
|
102
|
+
"ephemeral_5m_input_tokens": 0
|
|
103
|
+
},
|
|
104
|
+
"inference_geo": "",
|
|
105
|
+
"iterations": [],
|
|
106
|
+
"speed": "standard"
|
|
107
|
+
},
|
|
108
|
+
"permission_denials": [],
|
|
109
|
+
"modelUsage": {
|
|
110
|
+
"winter-test/echo": {
|
|
111
|
+
"inputTokens": "<inputTokens>",
|
|
112
|
+
"outputTokens": "<outputTokens>",
|
|
113
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
114
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
115
|
+
"webSearchRequests": 0,
|
|
116
|
+
"costUSD": 0,
|
|
117
|
+
"canonicalModel": "winter-test/echo"
|
|
118
|
+
}
|
|
119
|
+
}
|
|
86
120
|
}
|
|
87
121
|
}
|
|
88
122
|
]
|
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
"TaskOutput",
|
|
40
40
|
"TaskStop",
|
|
41
41
|
"TaskUpdate",
|
|
42
|
+
"WebFetch",
|
|
43
|
+
"WebSearch",
|
|
42
44
|
"Workflow",
|
|
43
45
|
"Write"
|
|
44
46
|
],
|
|
@@ -82,7 +84,39 @@
|
|
|
82
84
|
"subtype": "success",
|
|
83
85
|
"is_error": false,
|
|
84
86
|
"result": "[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"<system-reminder>\\nAvailable agent types for the Agent tool:\\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \\\"src/components/**/*.tsx\\\"), grep for symbols or keywords (eg. \\\"API endpoints\\\"), or answer \\\"where is X defined / which files reference Y.\\\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \\\"quick\\\" for a single targeted lookup, \\\"medium\\\" for moderate exploration, or \\\"very thorough\\\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n\\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\\n</system-reminder>\\n\"},{\"type\":\"text\",\"text\":\"<system-reminder>\\nAs you answer the user's questions, you can use the following context:\\n# currentDate\\nToday's date is <FIXTURE-DATE>.\\n\\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\\n</system-reminder>\\n\\n\"},{\"type\":\"text\",\"text\":\"first\"}]}]",
|
|
85
|
-
"
|
|
87
|
+
"usage": {
|
|
88
|
+
"output_tokens_details": {
|
|
89
|
+
"thinking_tokens": 0
|
|
90
|
+
},
|
|
91
|
+
"input_tokens": "<input_tokens>",
|
|
92
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
93
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
94
|
+
"output_tokens": "<output_tokens>",
|
|
95
|
+
"server_tool_use": {
|
|
96
|
+
"web_search_requests": 0,
|
|
97
|
+
"web_fetch_requests": 0
|
|
98
|
+
},
|
|
99
|
+
"service_tier": "standard",
|
|
100
|
+
"cache_creation": {
|
|
101
|
+
"ephemeral_1h_input_tokens": 0,
|
|
102
|
+
"ephemeral_5m_input_tokens": 0
|
|
103
|
+
},
|
|
104
|
+
"inference_geo": "",
|
|
105
|
+
"iterations": [],
|
|
106
|
+
"speed": "standard"
|
|
107
|
+
},
|
|
108
|
+
"permission_denials": [],
|
|
109
|
+
"modelUsage": {
|
|
110
|
+
"winter-test/echo": {
|
|
111
|
+
"inputTokens": "<inputTokens>",
|
|
112
|
+
"outputTokens": "<outputTokens>",
|
|
113
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
114
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
115
|
+
"webSearchRequests": 0,
|
|
116
|
+
"costUSD": 0,
|
|
117
|
+
"canonicalModel": "winter-test/echo"
|
|
118
|
+
}
|
|
119
|
+
}
|
|
86
120
|
}
|
|
87
121
|
},
|
|
88
122
|
{
|
|
@@ -125,6 +159,8 @@
|
|
|
125
159
|
"TaskOutput",
|
|
126
160
|
"TaskStop",
|
|
127
161
|
"TaskUpdate",
|
|
162
|
+
"WebFetch",
|
|
163
|
+
"WebSearch",
|
|
128
164
|
"Workflow",
|
|
129
165
|
"Write"
|
|
130
166
|
],
|
|
@@ -168,7 +204,39 @@
|
|
|
168
204
|
"subtype": "success",
|
|
169
205
|
"is_error": false,
|
|
170
206
|
"result": "[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"<system-reminder>\\nAvailable agent types for the Agent tool:\\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \\\"src/components/**/*.tsx\\\"), grep for symbols or keywords (eg. \\\"API endpoints\\\"), or answer \\\"where is X defined / which files reference Y.\\\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \\\"quick\\\" for a single targeted lookup, \\\"medium\\\" for moderate exploration, or \\\"very thorough\\\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\n\\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\\n</system-reminder>\\n\"},{\"type\":\"text\",\"text\":\"<system-reminder>\\nAs you answer the user's questions, you can use the following context:\\n# currentDate\\nToday's date is <FIXTURE-DATE>.\\n\\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\\n</system-reminder>\\n\\n\"},{\"type\":\"text\",\"text\":\"first\"}]},{\"role\":\"assistant\",\"content\":\"[{\\\"role\\\":\\\"user\\\",\\\"content\\\":[{\\\"type\\\":\\\"text\\\",\\\"text\\\":\\\"<system-reminder>\\\\nAvailable agent types for the Agent tool:\\\\n- claude: Catch-all for any task that doesn't fit a more specific agent. (Tools: *)\\\\n- Explore: Fast read-only search agent for locating code. Use it to find files by pattern (eg. \\\\\\\"src/components/**/*.tsx\\\\\\\"), grep for symbols or keywords (eg. \\\\\\\"API endpoints\\\\\\\"), or answer \\\\\\\"where is X defined / which files reference Y.\\\\\\\" Do NOT use it for code review, design-doc auditing, cross-file consistency checks, or open-ended analysis — it reads excerpts rather than whole files and will miss content past its read window. When calling, specify search breadth: \\\\\\\"quick\\\\\\\" for a single targeted lookup, \\\\\\\"medium\\\\\\\" for moderate exploration, or \\\\\\\"very thorough\\\\\\\" to search across multiple locations and naming conventions. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\\\n- general-purpose: General-purpose agent for researching complex questions, searching for code, and executing multi-step tasks. When you are searching for a keyword or file and are not confident that you will find the right match in the first few tries use this agent to perform the search for you. (Tools: *)\\\\n- Plan: Software architect agent for designing implementation plans. Use this when you need to plan the implementation strategy for a task. Returns step-by-step plans, identifies critical files, and considers architectural trade-offs. (Tools: All tools except Agent, Artifact, ExitPlanMode, Edit, Write, NotebookEdit)\\\\n\\\\nWhen you launch multiple agents for independent work, send them in a single message with multiple tool uses so they run concurrently.\\\\n</system-reminder>\\\\n\\\"},{\\\"type\\\":\\\"text\\\",\\\"text\\\":\\\"<system-reminder>\\\\nAs you answer the user's questions, you can use the following context:\\\\n# currentDate\\\\nToday's date is <FIXTURE-DATE>.\\\\n\\\\n IMPORTANT: this context may or may not be relevant to your tasks. You should not respond to this context unless it is highly relevant to your task.\\\\n</system-reminder>\\\\n\\\\n\\\"},{\\\"type\\\":\\\"text\\\",\\\"text\\\":\\\"first\\\"}]}]\"},{\"role\":\"user\",\"content\":\"second\"}]",
|
|
171
|
-
"
|
|
207
|
+
"usage": {
|
|
208
|
+
"output_tokens_details": {
|
|
209
|
+
"thinking_tokens": 0
|
|
210
|
+
},
|
|
211
|
+
"input_tokens": "<input_tokens>",
|
|
212
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
213
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
214
|
+
"output_tokens": "<output_tokens>",
|
|
215
|
+
"server_tool_use": {
|
|
216
|
+
"web_search_requests": 0,
|
|
217
|
+
"web_fetch_requests": 0
|
|
218
|
+
},
|
|
219
|
+
"service_tier": "standard",
|
|
220
|
+
"cache_creation": {
|
|
221
|
+
"ephemeral_1h_input_tokens": 0,
|
|
222
|
+
"ephemeral_5m_input_tokens": 0
|
|
223
|
+
},
|
|
224
|
+
"inference_geo": "",
|
|
225
|
+
"iterations": [],
|
|
226
|
+
"speed": "standard"
|
|
227
|
+
},
|
|
228
|
+
"permission_denials": [],
|
|
229
|
+
"modelUsage": {
|
|
230
|
+
"winter-test/echo": {
|
|
231
|
+
"inputTokens": "<inputTokens>",
|
|
232
|
+
"outputTokens": "<outputTokens>",
|
|
233
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
234
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
235
|
+
"webSearchRequests": 0,
|
|
236
|
+
"costUSD": 0,
|
|
237
|
+
"canonicalModel": "winter-test/echo"
|
|
238
|
+
}
|
|
239
|
+
}
|
|
172
240
|
}
|
|
173
241
|
}
|
|
174
242
|
]
|
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
"TaskOutput",
|
|
40
40
|
"TaskStop",
|
|
41
41
|
"TaskUpdate",
|
|
42
|
+
"WebFetch",
|
|
43
|
+
"WebSearch",
|
|
42
44
|
"Workflow",
|
|
43
45
|
"Write"
|
|
44
46
|
],
|
|
@@ -208,7 +210,39 @@
|
|
|
208
210
|
"subtype": "success",
|
|
209
211
|
"is_error": false,
|
|
210
212
|
"result": "messaging done",
|
|
211
|
-
"
|
|
213
|
+
"usage": {
|
|
214
|
+
"output_tokens_details": {
|
|
215
|
+
"thinking_tokens": 0
|
|
216
|
+
},
|
|
217
|
+
"input_tokens": "<input_tokens>",
|
|
218
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
219
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
220
|
+
"output_tokens": "<output_tokens>",
|
|
221
|
+
"server_tool_use": {
|
|
222
|
+
"web_search_requests": 0,
|
|
223
|
+
"web_fetch_requests": 0
|
|
224
|
+
},
|
|
225
|
+
"service_tier": "standard",
|
|
226
|
+
"cache_creation": {
|
|
227
|
+
"ephemeral_1h_input_tokens": 0,
|
|
228
|
+
"ephemeral_5m_input_tokens": 0
|
|
229
|
+
},
|
|
230
|
+
"inference_geo": "",
|
|
231
|
+
"iterations": [],
|
|
232
|
+
"speed": "standard"
|
|
233
|
+
},
|
|
234
|
+
"permission_denials": [],
|
|
235
|
+
"modelUsage": {
|
|
236
|
+
"winter-test/echo": {
|
|
237
|
+
"inputTokens": "<inputTokens>",
|
|
238
|
+
"outputTokens": "<outputTokens>",
|
|
239
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
240
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
241
|
+
"webSearchRequests": 0,
|
|
242
|
+
"costUSD": 0,
|
|
243
|
+
"canonicalModel": "winter-test/echo"
|
|
244
|
+
}
|
|
245
|
+
}
|
|
212
246
|
}
|
|
213
247
|
}
|
|
214
248
|
]
|
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
"TaskOutput",
|
|
40
40
|
"TaskStop",
|
|
41
41
|
"TaskUpdate",
|
|
42
|
+
"WebFetch",
|
|
43
|
+
"WebSearch",
|
|
42
44
|
"Workflow",
|
|
43
45
|
"Write"
|
|
44
46
|
],
|
|
@@ -92,7 +94,7 @@
|
|
|
92
94
|
{
|
|
93
95
|
"type": "tool_result",
|
|
94
96
|
"tool_use_id": "p5-skill-1",
|
|
95
|
-
"content": "
|
|
97
|
+
"content": "Base directory for this skill: /winter-home/skills/p5probe\n\nP5 SKILL BODY MARKER\n"
|
|
96
98
|
}
|
|
97
99
|
]
|
|
98
100
|
}
|
|
@@ -123,7 +125,39 @@
|
|
|
123
125
|
"subtype": "success",
|
|
124
126
|
"is_error": false,
|
|
125
127
|
"result": "skill done",
|
|
126
|
-
"
|
|
128
|
+
"usage": {
|
|
129
|
+
"output_tokens_details": {
|
|
130
|
+
"thinking_tokens": 0
|
|
131
|
+
},
|
|
132
|
+
"input_tokens": "<input_tokens>",
|
|
133
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
134
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
135
|
+
"output_tokens": "<output_tokens>",
|
|
136
|
+
"server_tool_use": {
|
|
137
|
+
"web_search_requests": 0,
|
|
138
|
+
"web_fetch_requests": 0
|
|
139
|
+
},
|
|
140
|
+
"service_tier": "standard",
|
|
141
|
+
"cache_creation": {
|
|
142
|
+
"ephemeral_1h_input_tokens": 0,
|
|
143
|
+
"ephemeral_5m_input_tokens": 0
|
|
144
|
+
},
|
|
145
|
+
"inference_geo": "",
|
|
146
|
+
"iterations": [],
|
|
147
|
+
"speed": "standard"
|
|
148
|
+
},
|
|
149
|
+
"permission_denials": [],
|
|
150
|
+
"modelUsage": {
|
|
151
|
+
"winter-test/echo": {
|
|
152
|
+
"inputTokens": "<inputTokens>",
|
|
153
|
+
"outputTokens": "<outputTokens>",
|
|
154
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
155
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
156
|
+
"webSearchRequests": 0,
|
|
157
|
+
"costUSD": 0,
|
|
158
|
+
"canonicalModel": "winter-test/echo"
|
|
159
|
+
}
|
|
160
|
+
}
|
|
127
161
|
}
|
|
128
162
|
}
|
|
129
163
|
]
|
|
@@ -40,6 +40,8 @@
|
|
|
40
40
|
"TaskOutput",
|
|
41
41
|
"TaskStop",
|
|
42
42
|
"TaskUpdate",
|
|
43
|
+
"WebFetch",
|
|
44
|
+
"WebSearch",
|
|
43
45
|
"Workflow",
|
|
44
46
|
"Write"
|
|
45
47
|
],
|
|
@@ -182,7 +184,39 @@
|
|
|
182
184
|
"is_error": true,
|
|
183
185
|
"result": "Failed to provide valid structured output after 3 attempts",
|
|
184
186
|
"terminal_reason": "structured_output_retry_exhausted",
|
|
185
|
-
"
|
|
187
|
+
"usage": {
|
|
188
|
+
"output_tokens_details": {
|
|
189
|
+
"thinking_tokens": 0
|
|
190
|
+
},
|
|
191
|
+
"input_tokens": "<input_tokens>",
|
|
192
|
+
"cache_creation_input_tokens": "<cache_creation_input_tokens>",
|
|
193
|
+
"cache_read_input_tokens": "<cache_read_input_tokens>",
|
|
194
|
+
"output_tokens": "<output_tokens>",
|
|
195
|
+
"server_tool_use": {
|
|
196
|
+
"web_search_requests": 0,
|
|
197
|
+
"web_fetch_requests": 0
|
|
198
|
+
},
|
|
199
|
+
"service_tier": "standard",
|
|
200
|
+
"cache_creation": {
|
|
201
|
+
"ephemeral_1h_input_tokens": 0,
|
|
202
|
+
"ephemeral_5m_input_tokens": 0
|
|
203
|
+
},
|
|
204
|
+
"inference_geo": "",
|
|
205
|
+
"iterations": [],
|
|
206
|
+
"speed": "standard"
|
|
207
|
+
},
|
|
208
|
+
"permission_denials": [],
|
|
209
|
+
"modelUsage": {
|
|
210
|
+
"winter-test/echo": {
|
|
211
|
+
"inputTokens": "<inputTokens>",
|
|
212
|
+
"outputTokens": "<outputTokens>",
|
|
213
|
+
"cacheReadInputTokens": "<cacheReadInputTokens>",
|
|
214
|
+
"cacheCreationInputTokens": "<cacheCreationInputTokens>",
|
|
215
|
+
"webSearchRequests": 0,
|
|
216
|
+
"costUSD": 0,
|
|
217
|
+
"canonicalModel": "winter-test/echo"
|
|
218
|
+
}
|
|
219
|
+
}
|
|
186
220
|
}
|
|
187
221
|
}
|
|
188
222
|
]
|