@doist/todoist-mcp 12.5.4 → 12.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +25 -27
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -3
- package/dist/main-http.js +2 -2
- package/dist/main.js +1 -1
- package/dist/{mcp-server-BxS0NkIc.js → mcp-server-Drv9wBc-.js} +966 -963
- package/dist/mcp-server.d.ts +1 -1
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/{require-valid-todoist-token-D0hjlcRK.js → require-valid-todoist-token-CC6hWyEx.js} +1 -1
- package/dist/tool-registry.d.ts +25 -27
- package/dist/tool-registry.d.ts.map +1 -1
- package/dist/tools/find-activity.d.ts.map +1 -1
- package/dist/tools/get-overview.d.ts +4 -4
- package/dist/tools/get-overview.d.ts.map +1 -1
- package/dist/tools/get-project-activity-stats.d.ts +2 -2
- package/dist/tools/get-project-health.d.ts +17 -17
- package/dist/tools/get-project-health.d.ts.map +1 -1
- package/dist/tools/get-workspace-insights.d.ts +4 -6
- package/dist/tools/get-workspace-insights.d.ts.map +1 -1
- package/package.json +1 -1
- package/scripts/eval-instructions.ts +150 -42
|
@@ -14,9 +14,11 @@
|
|
|
14
14
|
* git stash pop
|
|
15
15
|
* npx tsx scripts/eval-instructions.ts --label after
|
|
16
16
|
*
|
|
17
|
-
* Tool calls are never executed
|
|
18
|
-
*
|
|
19
|
-
*
|
|
17
|
+
* Tool calls are never executed against Todoist, so this touches no Todoist
|
|
18
|
+
* data and needs no Todoist token. Judging looks at one call per attempt: the
|
|
19
|
+
* first, unless it is a context tool (see CONTEXT_TOOLS), in which case a
|
|
20
|
+
* canned result is fed back and the next call is judged. It does spend money
|
|
21
|
+
* on model calls; see --repeats and --models.
|
|
20
22
|
*
|
|
21
23
|
* Auth uses the Anthropic SDK's standard credential chain: either
|
|
22
24
|
* ANTHROPIC_API_KEY (a key from the Anthropic Console at platform.claude.com),
|
|
@@ -34,9 +36,17 @@ import Anthropic from '@anthropic-ai/sdk'
|
|
|
34
36
|
import { z } from 'zod'
|
|
35
37
|
import { instructions } from '../src/mcp-server.js'
|
|
36
38
|
import { registeredTools } from '../src/tool-registry.js'
|
|
39
|
+
import type { UserInfoStructured } from '../src/tools/user-info.js'
|
|
37
40
|
import { createLimiter } from '../src/utils/concurrency.js'
|
|
38
41
|
import { ToolNames } from '../src/utils/tool-names.js'
|
|
39
42
|
|
|
43
|
+
/**
|
|
44
|
+
* The user ID the stubbed `user-info` reports. A scenario asking a first-person
|
|
45
|
+
* question can assert the model carried it into a filter argument rather than
|
|
46
|
+
* querying everyone.
|
|
47
|
+
*/
|
|
48
|
+
const EVAL_USER_ID = '2671355'
|
|
49
|
+
|
|
40
50
|
type Check = (input: Record<string, unknown>) => string | null
|
|
41
51
|
|
|
42
52
|
/** First item of a batch tool's array argument, e.g. add-tasks' `tasks`. */
|
|
@@ -89,9 +99,14 @@ const SCENARIOS: Scenario[] = [
|
|
|
89
99
|
if (!input.dateFrom || !input.dateTo) {
|
|
90
100
|
return 'no dateFrom/dateTo, so this searches all history rather than last week'
|
|
91
101
|
}
|
|
102
|
+
// find-activity reports every user's events by default, so "what did
|
|
103
|
+
// I get done" answered without this includes collaborators' work.
|
|
104
|
+
if (input.initiatorId !== EVAL_USER_ID) {
|
|
105
|
+
return `initiatorId is ${JSON.stringify(input.initiatorId)}, so this reports every collaborator's completions, not "I"`
|
|
106
|
+
}
|
|
92
107
|
return null
|
|
93
108
|
},
|
|
94
|
-
guards: 'instructions: find-activity vs find-completed-tasks, over a bounded range',
|
|
109
|
+
guards: 'instructions: find-activity vs find-completed-tasks, over a bounded range, filtered to the asker',
|
|
95
110
|
},
|
|
96
111
|
{
|
|
97
112
|
id: 'resolve-person',
|
|
@@ -195,6 +210,51 @@ const DEFAULT_MODELS = ['claude-haiku-4-5', 'claude-sonnet-5']
|
|
|
195
210
|
const DEFAULT_REPEATS = 5
|
|
196
211
|
const MAX_CONCURRENCY = 4
|
|
197
212
|
|
|
213
|
+
/**
|
|
214
|
+
* Tools a model may legitimately call before the one a scenario is testing, to
|
|
215
|
+
* establish context it has no other way to get — today's date, the user's
|
|
216
|
+
* timezone. "What did I get done last week?" cannot be turned into a date range
|
|
217
|
+
* without them.
|
|
218
|
+
*
|
|
219
|
+
* Judging the very first call outright scores that correct behaviour as a
|
|
220
|
+
* failure: an `expect` allowlist naming the tool under test cannot also name
|
|
221
|
+
* every reasonable lookup that precedes it. So a call to one of these is
|
|
222
|
+
* answered with the canned result below and judging moves to the next call.
|
|
223
|
+
*
|
|
224
|
+
* The result is fiction, but its shape has to be real: a scenario may assert
|
|
225
|
+
* that the model carried a value through (see EVAL_USER_ID), and a model reads
|
|
226
|
+
* a malformed field the way it would read any other bad data. `satisfies` ties
|
|
227
|
+
* it to the tool's own output type, so a renamed or added field breaks here
|
|
228
|
+
* rather than drifting quietly. Formats must match too — `currentLocalTime`
|
|
229
|
+
* is what `toLocaleString('en-US', …)` produces, not ISO.
|
|
230
|
+
*/
|
|
231
|
+
const CONTEXT_TOOLS: Record<string, unknown> = {
|
|
232
|
+
[ToolNames.USER_INFO]: {
|
|
233
|
+
type: 'user_info',
|
|
234
|
+
userId: EVAL_USER_ID,
|
|
235
|
+
fullName: 'Eval User',
|
|
236
|
+
timezone: 'Europe/London',
|
|
237
|
+
currentLocalTime: '08/11/2026, 09:30:00',
|
|
238
|
+
startDay: 1,
|
|
239
|
+
startDayName: 'Monday',
|
|
240
|
+
weekStartDate: '2026-08-10',
|
|
241
|
+
weekEndDate: '2026-08-16',
|
|
242
|
+
currentWeekNumber: 33,
|
|
243
|
+
completedToday: 3,
|
|
244
|
+
dailyGoal: 5,
|
|
245
|
+
weeklyGoal: 30,
|
|
246
|
+
email: 'eval@example.com',
|
|
247
|
+
plan: 'Todoist Pro',
|
|
248
|
+
} satisfies UserInfoStructured,
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
/**
|
|
252
|
+
* How many context calls an attempt may make before it is judged a failure.
|
|
253
|
+
* A model that keeps gathering context is not answering the question, and
|
|
254
|
+
* without a cap a loop would bill for turns forever.
|
|
255
|
+
*/
|
|
256
|
+
const MAX_CONTEXT_HOPS = 3
|
|
257
|
+
|
|
198
258
|
function parseArgs() {
|
|
199
259
|
const args = process.argv.slice(2)
|
|
200
260
|
const get = (flag: string) => {
|
|
@@ -254,6 +314,28 @@ type Attempt = {
|
|
|
254
314
|
*/
|
|
255
315
|
errored: boolean
|
|
256
316
|
usage: Usage
|
|
317
|
+
/** Context calls answered with a stub before the judged one. */
|
|
318
|
+
contextHops: number
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
/** Verdict on the one call a scenario is judged by. */
|
|
322
|
+
function judge(
|
|
323
|
+
scenario: Scenario,
|
|
324
|
+
call: Anthropic.ToolUseBlock,
|
|
325
|
+
): { pass: boolean; reason: string } {
|
|
326
|
+
if (scenario.forbid?.includes(call.name)) {
|
|
327
|
+
return { pass: false, reason: `called ${call.name}, which this rule forbids` }
|
|
328
|
+
}
|
|
329
|
+
if (scenario.expect && !scenario.expect.includes(call.name)) {
|
|
330
|
+
return { pass: false, reason: `expected ${scenario.expect.join(' or ')}` }
|
|
331
|
+
}
|
|
332
|
+
// An argument check written for a specific tool must not run against a
|
|
333
|
+
// different one a forbid-only scenario legitimately allows.
|
|
334
|
+
if (scenario.forbid && !scenario.expect) {
|
|
335
|
+
return { pass: true, reason: '' }
|
|
336
|
+
}
|
|
337
|
+
const reason = scenario.check?.(call.input as Record<string, unknown>) ?? null
|
|
338
|
+
return { pass: reason === null, reason: reason ?? '' }
|
|
257
339
|
}
|
|
258
340
|
|
|
259
341
|
async function runAttempt(
|
|
@@ -267,51 +349,77 @@ async function runAttempt(
|
|
|
267
349
|
model,
|
|
268
350
|
errored: false,
|
|
269
351
|
usage: { input: 0, output: 0, cacheWrite: 0, cacheRead: 0 },
|
|
352
|
+
contextHops: 0,
|
|
270
353
|
}
|
|
354
|
+
const messages: Anthropic.MessageParam[] = [{ role: 'user', content: scenario.prompt }]
|
|
271
355
|
try {
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
356
|
+
for (let hop = 0; ; hop++) {
|
|
357
|
+
const response = await client.messages.create({
|
|
358
|
+
model,
|
|
359
|
+
max_tokens: 4096,
|
|
360
|
+
// Tools render before system, so one breakpoint here caches both.
|
|
361
|
+
system: [
|
|
362
|
+
{ type: 'text', text: instructions, cache_control: { type: 'ephemeral' } },
|
|
363
|
+
],
|
|
364
|
+
tools,
|
|
365
|
+
messages,
|
|
366
|
+
})
|
|
280
367
|
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
368
|
+
base.usage = {
|
|
369
|
+
input: base.usage.input + response.usage.input_tokens,
|
|
370
|
+
output: base.usage.output + response.usage.output_tokens,
|
|
371
|
+
cacheWrite:
|
|
372
|
+
base.usage.cacheWrite + (response.usage.cache_creation_input_tokens ?? 0),
|
|
373
|
+
cacheRead: base.usage.cacheRead + (response.usage.cache_read_input_tokens ?? 0),
|
|
374
|
+
}
|
|
287
375
|
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
if (scenario.forbid?.includes(call.name)) {
|
|
293
|
-
return {
|
|
294
|
-
...base,
|
|
295
|
-
calledTool: call.name,
|
|
296
|
-
pass: false,
|
|
297
|
-
reason: `called ${call.name}, which this rule forbids`,
|
|
376
|
+
const calls = response.content.filter((b) => b.type === 'tool_use')
|
|
377
|
+
const call = calls[0]
|
|
378
|
+
if (!call) {
|
|
379
|
+
return { ...base, calledTool: null, pass: false, reason: 'no tool call' }
|
|
298
380
|
}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
381
|
+
|
|
382
|
+
// A turn can carry several calls at once. Judge the substantive one
|
|
383
|
+
// rather than whichever block came first: a model that asks for the
|
|
384
|
+
// date and queries the activity log in the same turn has made its
|
|
385
|
+
// choice, and the order between the two is arbitrary. A forbidden
|
|
386
|
+
// call outranks that — catching it is the point of a forbid rule.
|
|
387
|
+
//
|
|
388
|
+
// A tool a scenario expects is never treated as context, or a
|
|
389
|
+
// scenario testing the route *to* user-info could never pass: its
|
|
390
|
+
// expected call would be stubbed instead of judged.
|
|
391
|
+
const isContext = (c: Anthropic.ToolUseBlock) =>
|
|
392
|
+
Object.hasOwn(CONTEXT_TOOLS, c.name) && !scenario.expect?.includes(c.name)
|
|
393
|
+
const substantive =
|
|
394
|
+
calls.find((c) => scenario.forbid?.includes(c.name)) ??
|
|
395
|
+
calls.find((c) => !isContext(c))
|
|
396
|
+
if (substantive) {
|
|
397
|
+
const { pass, reason } = judge(scenario, substantive)
|
|
398
|
+
return { ...base, calledTool: substantive.name, pass, reason: reason || null }
|
|
306
399
|
}
|
|
400
|
+
|
|
401
|
+
// Nothing but context gathering, so answer it and judge what the
|
|
402
|
+
// model reaches for next.
|
|
403
|
+
if (hop >= MAX_CONTEXT_HOPS) {
|
|
404
|
+
return {
|
|
405
|
+
...base,
|
|
406
|
+
calledTool: call.name,
|
|
407
|
+
pass: false,
|
|
408
|
+
reason: `still gathering context after ${MAX_CONTEXT_HOPS} calls`,
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
base.contextHops = hop + 1
|
|
413
|
+
messages.push({ role: 'assistant', content: response.content })
|
|
414
|
+
messages.push({
|
|
415
|
+
role: 'user',
|
|
416
|
+
content: calls.map((c) => ({
|
|
417
|
+
type: 'tool_result' as const,
|
|
418
|
+
tool_use_id: c.id,
|
|
419
|
+
content: JSON.stringify(CONTEXT_TOOLS[c.name]),
|
|
420
|
+
})),
|
|
421
|
+
})
|
|
307
422
|
}
|
|
308
|
-
// An argument check written for a specific tool must not run against a
|
|
309
|
-
// different one a forbid-only scenario legitimately allows.
|
|
310
|
-
if (scenario.forbid && !scenario.expect) {
|
|
311
|
-
return { ...base, calledTool: call.name, pass: true, reason: null }
|
|
312
|
-
}
|
|
313
|
-
const reason = scenario.check?.(call.input as Record<string, unknown>) ?? null
|
|
314
|
-
return { ...base, calledTool: call.name, pass: reason === null, reason }
|
|
315
423
|
} catch (error) {
|
|
316
424
|
const message = error instanceof Error ? error.message : String(error)
|
|
317
425
|
return {
|