@doist/todoist-mcp 12.5.4 → 12.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,9 +14,11 @@
14
14
  * git stash pop
15
15
  * npx tsx scripts/eval-instructions.ts --label after
16
16
  *
17
- * Tool calls are never executed — only the first call of each turn is
18
- * inspected — so this touches no Todoist data and needs no Todoist token. It
19
- * does spend money on model calls; see --repeats and --models.
17
+ * Tool calls are never executed against Todoist, so this touches no Todoist
18
+ * data and needs no Todoist token. Judging looks at one call per attempt: the
19
+ * first, unless it is a context tool (see CONTEXT_TOOLS), in which case a
20
+ * canned result is fed back and the next call is judged. It does spend money
21
+ * on model calls; see --repeats and --models.
20
22
  *
21
23
  * Auth uses the Anthropic SDK's standard credential chain: either
22
24
  * ANTHROPIC_API_KEY (a key from the Anthropic Console at platform.claude.com),
@@ -34,9 +36,17 @@ import Anthropic from '@anthropic-ai/sdk'
34
36
  import { z } from 'zod'
35
37
  import { instructions } from '../src/mcp-server.js'
36
38
  import { registeredTools } from '../src/tool-registry.js'
39
+ import type { UserInfoStructured } from '../src/tools/user-info.js'
37
40
  import { createLimiter } from '../src/utils/concurrency.js'
38
41
  import { ToolNames } from '../src/utils/tool-names.js'
39
42
 
43
+ /**
44
+ * The user ID the stubbed `user-info` reports. A scenario asking a first-person
45
+ * question can assert the model carried it into a filter argument rather than
46
+ * querying everyone.
47
+ */
48
+ const EVAL_USER_ID = '2671355'
49
+
40
50
  type Check = (input: Record<string, unknown>) => string | null
41
51
 
42
52
  /** First item of a batch tool's array argument, e.g. add-tasks' `tasks`. */
@@ -89,9 +99,14 @@ const SCENARIOS: Scenario[] = [
89
99
  if (!input.dateFrom || !input.dateTo) {
90
100
  return 'no dateFrom/dateTo, so this searches all history rather than last week'
91
101
  }
102
+ // find-activity reports every user's events by default, so "what did
103
+ // I get done" answered without this includes collaborators' work.
104
+ if (input.initiatorId !== EVAL_USER_ID) {
105
+ return `initiatorId is ${JSON.stringify(input.initiatorId)}, so this reports every collaborator's completions, not "I"`
106
+ }
92
107
  return null
93
108
  },
94
- guards: 'instructions: find-activity vs find-completed-tasks, over a bounded range',
109
+ guards: 'instructions: find-activity vs find-completed-tasks, over a bounded range, filtered to the asker',
95
110
  },
96
111
  {
97
112
  id: 'resolve-person',
@@ -195,6 +210,51 @@ const DEFAULT_MODELS = ['claude-haiku-4-5', 'claude-sonnet-5']
195
210
  const DEFAULT_REPEATS = 5
196
211
  const MAX_CONCURRENCY = 4
197
212
 
213
+ /**
214
+ * Tools a model may legitimately call before the one a scenario is testing, to
215
+ * establish context it has no other way to get — today's date, the user's
216
+ * timezone. "What did I get done last week?" cannot be turned into a date range
217
+ * without them.
218
+ *
219
+ * Judging the very first call outright scores that correct behaviour as a
220
+ * failure: an `expect` allowlist naming the tool under test cannot also name
221
+ * every reasonable lookup that precedes it. So a call to one of these is
222
+ * answered with the canned result below and judging moves to the next call.
223
+ *
224
+ * The result is fiction, but its shape has to be real: a scenario may assert
225
+ * that the model carried a value through (see EVAL_USER_ID), and a model reads
226
+ * a malformed field the way it would read any other bad data. `satisfies` ties
227
+ * it to the tool's own output type, so a renamed or added field breaks here
228
+ * rather than drifting quietly. Formats must match too — `currentLocalTime`
229
+ * is what `toLocaleString('en-US', …)` produces, not ISO.
230
+ */
231
+ const CONTEXT_TOOLS: Record<string, unknown> = {
232
+ [ToolNames.USER_INFO]: {
233
+ type: 'user_info',
234
+ userId: EVAL_USER_ID,
235
+ fullName: 'Eval User',
236
+ timezone: 'Europe/London',
237
+ currentLocalTime: '08/11/2026, 09:30:00',
238
+ startDay: 1,
239
+ startDayName: 'Monday',
240
+ weekStartDate: '2026-08-10',
241
+ weekEndDate: '2026-08-16',
242
+ currentWeekNumber: 33,
243
+ completedToday: 3,
244
+ dailyGoal: 5,
245
+ weeklyGoal: 30,
246
+ email: 'eval@example.com',
247
+ plan: 'Todoist Pro',
248
+ } satisfies UserInfoStructured,
249
+ }
250
+
251
+ /**
252
+ * How many context calls an attempt may make before it is judged a failure.
253
+ * A model that keeps gathering context is not answering the question, and
254
+ * without a cap a loop would bill for turns forever.
255
+ */
256
+ const MAX_CONTEXT_HOPS = 3
257
+
198
258
  function parseArgs() {
199
259
  const args = process.argv.slice(2)
200
260
  const get = (flag: string) => {
@@ -254,6 +314,28 @@ type Attempt = {
254
314
  */
255
315
  errored: boolean
256
316
  usage: Usage
317
+ /** Context calls answered with a stub before the judged one. */
318
+ contextHops: number
319
+ }
320
+
321
+ /** Verdict on the one call a scenario is judged by. */
322
+ function judge(
323
+ scenario: Scenario,
324
+ call: Anthropic.ToolUseBlock,
325
+ ): { pass: boolean; reason: string } {
326
+ if (scenario.forbid?.includes(call.name)) {
327
+ return { pass: false, reason: `called ${call.name}, which this rule forbids` }
328
+ }
329
+ if (scenario.expect && !scenario.expect.includes(call.name)) {
330
+ return { pass: false, reason: `expected ${scenario.expect.join(' or ')}` }
331
+ }
332
+ // An argument check written for a specific tool must not run against a
333
+ // different one a forbid-only scenario legitimately allows.
334
+ if (scenario.forbid && !scenario.expect) {
335
+ return { pass: true, reason: '' }
336
+ }
337
+ const reason = scenario.check?.(call.input as Record<string, unknown>) ?? null
338
+ return { pass: reason === null, reason: reason ?? '' }
257
339
  }
258
340
 
259
341
  async function runAttempt(
@@ -267,51 +349,77 @@ async function runAttempt(
267
349
  model,
268
350
  errored: false,
269
351
  usage: { input: 0, output: 0, cacheWrite: 0, cacheRead: 0 },
352
+ contextHops: 0,
270
353
  }
354
+ const messages: Anthropic.MessageParam[] = [{ role: 'user', content: scenario.prompt }]
271
355
  try {
272
- const response = await client.messages.create({
273
- model,
274
- max_tokens: 4096,
275
- // Tools render before system, so one breakpoint here caches both.
276
- system: [{ type: 'text', text: instructions, cache_control: { type: 'ephemeral' } }],
277
- tools,
278
- messages: [{ role: 'user', content: scenario.prompt }],
279
- })
356
+ for (let hop = 0; ; hop++) {
357
+ const response = await client.messages.create({
358
+ model,
359
+ max_tokens: 4096,
360
+ // Tools render before system, so one breakpoint here caches both.
361
+ system: [
362
+ { type: 'text', text: instructions, cache_control: { type: 'ephemeral' } },
363
+ ],
364
+ tools,
365
+ messages,
366
+ })
280
367
 
281
- base.usage = {
282
- input: response.usage.input_tokens,
283
- output: response.usage.output_tokens,
284
- cacheWrite: response.usage.cache_creation_input_tokens ?? 0,
285
- cacheRead: response.usage.cache_read_input_tokens ?? 0,
286
- }
368
+ base.usage = {
369
+ input: base.usage.input + response.usage.input_tokens,
370
+ output: base.usage.output + response.usage.output_tokens,
371
+ cacheWrite:
372
+ base.usage.cacheWrite + (response.usage.cache_creation_input_tokens ?? 0),
373
+ cacheRead: base.usage.cacheRead + (response.usage.cache_read_input_tokens ?? 0),
374
+ }
287
375
 
288
- const call = response.content.find((b) => b.type === 'tool_use')
289
- if (!call) {
290
- return { ...base, calledTool: null, pass: false, reason: 'no tool call' }
291
- }
292
- if (scenario.forbid?.includes(call.name)) {
293
- return {
294
- ...base,
295
- calledTool: call.name,
296
- pass: false,
297
- reason: `called ${call.name}, which this rule forbids`,
376
+ const calls = response.content.filter((b) => b.type === 'tool_use')
377
+ const call = calls[0]
378
+ if (!call) {
379
+ return { ...base, calledTool: null, pass: false, reason: 'no tool call' }
298
380
  }
299
- }
300
- if (scenario.expect && !scenario.expect.includes(call.name)) {
301
- return {
302
- ...base,
303
- calledTool: call.name,
304
- pass: false,
305
- reason: `expected ${scenario.expect.join(' or ')}`,
381
+
382
+ // A turn can carry several calls at once. Judge the substantive one
383
+ // rather than whichever block came first: a model that asks for the
384
+ // date and queries the activity log in the same turn has made its
385
+ // choice, and the order between the two is arbitrary. A forbidden
386
+ // call outranks that — catching it is the point of a forbid rule.
387
+ //
388
+ // A tool a scenario expects is never treated as context, or a
389
+ // scenario testing the route *to* user-info could never pass: its
390
+ // expected call would be stubbed instead of judged.
391
+ const isContext = (c: Anthropic.ToolUseBlock) =>
392
+ Object.hasOwn(CONTEXT_TOOLS, c.name) && !scenario.expect?.includes(c.name)
393
+ const substantive =
394
+ calls.find((c) => scenario.forbid?.includes(c.name)) ??
395
+ calls.find((c) => !isContext(c))
396
+ if (substantive) {
397
+ const { pass, reason } = judge(scenario, substantive)
398
+ return { ...base, calledTool: substantive.name, pass, reason: reason || null }
306
399
  }
400
+
401
+ // Nothing but context gathering, so answer it and judge what the
402
+ // model reaches for next.
403
+ if (hop >= MAX_CONTEXT_HOPS) {
404
+ return {
405
+ ...base,
406
+ calledTool: call.name,
407
+ pass: false,
408
+ reason: `still gathering context after ${MAX_CONTEXT_HOPS} calls`,
409
+ }
410
+ }
411
+
412
+ base.contextHops = hop + 1
413
+ messages.push({ role: 'assistant', content: response.content })
414
+ messages.push({
415
+ role: 'user',
416
+ content: calls.map((c) => ({
417
+ type: 'tool_result' as const,
418
+ tool_use_id: c.id,
419
+ content: JSON.stringify(CONTEXT_TOOLS[c.name]),
420
+ })),
421
+ })
307
422
  }
308
- // An argument check written for a specific tool must not run against a
309
- // different one a forbid-only scenario legitimately allows.
310
- if (scenario.forbid && !scenario.expect) {
311
- return { ...base, calledTool: call.name, pass: true, reason: null }
312
- }
313
- const reason = scenario.check?.(call.input as Record<string, unknown>) ?? null
314
- return { ...base, calledTool: call.name, pass: reason === null, reason }
315
423
  } catch (error) {
316
424
  const message = error instanceof Error ? error.message : String(error)
317
425
  return {