@doist/todoist-mcp 12.5.5 → 12.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +0 -8
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -3
- package/dist/main-http.js +2 -2
- package/dist/main.js +1 -1
- package/dist/{mcp-server-Cm72B6O_.js → mcp-server-HzS0zbN7.js} +471 -479
- package/dist/mcp-server.d.ts +1 -1
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/{require-valid-todoist-token-DFE6h87D.js → require-valid-todoist-token-B8Dy2Kgy.js} +1 -1
- package/dist/tool-registry.d.ts +0 -8
- package/dist/tool-registry.d.ts.map +1 -1
- package/dist/tools/find-activity.d.ts.map +1 -1
- package/dist/tools/get-project-health.d.ts +0 -8
- package/dist/tools/get-project-health.d.ts.map +1 -1
- package/package.json +2 -2
- package/scripts/eval-instructions.ts +150 -42
package/dist/mcp-server.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js';
|
|
2
2
|
import { FEATURE_NAMES, Feature, FeatureName, Features } from './mcp-helpers.js';
|
|
3
|
-
export declare const instructions = "\n## Todoist Task and Project Management Tools\n\nYou have access to comprehensive Todoist management tools for personal productivity and team collaboration. Use these tools to help users manage tasks, projects, sections, comments, and assignments effectively.\n\n### Core Capabilities:\n- Create, update, complete, and search tasks with rich metadata (priorities, due dates, durations, assignments)\n- Manage projects and sections with flexible organization\n- Handle comments and collaboration features\n- Bulk assignment operations for team workflows\n- Get overviews and insights about workload and progress\n\n### Choosing between tools\n\nWhat each tool does and how to fill its parameters is in the tool's own description and input schema. This section covers only what those cannot: which tool to reach for, and how tools relate to each other.\n\n**Dates**\n\n- To move a task to a different date use **reschedule-tasks**, never **update-tasks**. update-tasks replaces the whole due string, which destroys recurrence on recurring tasks.\n- Never send a task's existing projectId, sectionId or parentId back to **update-tasks** \u2014 those fields are treated as a move.\n- All dates respect the user's timezone.\n\n**Finding things**\n\n- For \"what did I complete?\",
|
|
3
|
+
export declare const instructions = "\n## Todoist Task and Project Management Tools\n\nYou have access to comprehensive Todoist management tools for personal productivity and team collaboration. Use these tools to help users manage tasks, projects, sections, comments, and assignments effectively.\n\n### Core Capabilities:\n- Create, update, complete, and search tasks with rich metadata (priorities, due dates, durations, assignments)\n- Manage projects and sections with flexible organization\n- Handle comments and collaboration features\n- Bulk assignment operations for team workflows\n- Get overviews and insights about workload and progress\n\n### Choosing between tools\n\nWhat each tool does and how to fill its parameters is in the tool's own description and input schema. This section covers only what those cannot: which tool to reach for, and how tools relate to each other.\n\n**Dates**\n\n- To move a task to a different date use **reschedule-tasks**, never **update-tasks**. update-tasks replaces the whole due string, which destroys recurrence on recurring tasks.\n- Never send a task's existing projectId, sectionId or parentId back to **update-tasks** \u2014 those fields are treated as a move.\n- All dates respect the user's timezone.\n\n**Finding things**\n\n- For \"what did I complete?\", call **user-info** for the asker's user ID, then **find-activity** with objectType=\"task\", eventType=\"completed\", initiatorId, and dateFrom/dateTo. Without initiatorId the answer covers every collaborator. find-activity reports completion events, including each occurrence of a recurring task; **find-completed-tasks** lists completed tasks, which is not the same question.\n- To resolve a person's name or email to a user ID, or to answer \"who is X?\", use **find-project-collaborators** with just a searchTerm. It covers every shared project you can access plus yourself, so an empty result means they collaborate on none of them \u2014 not that they do not exist.\n- Before assigning work to someone, call **find-project-collaborators** again with the target projectId. Resolving an ID only proves they collaborate on *some* project you can see; assignment fails unless they collaborate on that one.\n- To find out whether a task hides subtasks, use **fetch-object** with includeChildren rather than a speculative **find-tasks** call.\n- Filter tasks by label **name**. Label IDs are only for **delete-object** and **update-labels**. Shared labels can be renamed but not recoloured, reordered or favourited.\n\n**Deleting and archiving**\n\n- **delete-object** removes every object type; there is no per-type delete tool. Reminders use type \"reminder\", location reminders \"location_reminder\".\n- A workspace project must be archived with **project-management** before it can be deleted. A personal project can be deleted directly.\n\n**Templates**\n\n- Templates cannot be listed through this server, so only use an ID or URL the user gave you. To start a new project from one, call **add-projects** first and import into it. Imports write immediately and cannot be undone.\n\n**Project health**\n\n- Health data may be stale \u2014 check the isStale flag, and call **analyze-project-health** to refresh it before reading again.\n\n**Batches**\n\n- Prefer a batch tool over one call per item.\n- Where a batch tool returns per-item `failures` alongside successes, one failure does not undo the rest: never retry the whole batch, re-send only the items whose failure reason is fixable. **reschedule-tasks** is the exception \u2014 it has no per-item failures and throws if any task fails, so retry it as a whole.\n\nUse the **productivity-analysis** prompt for a combined productivity review \u2014 it pulls together user-info, get-productivity-stats and find-completed-tasks.\n\nAlways provide clear, actionable task titles and descriptions. Use the overview tools to give users context about their workload and project status.\n";
|
|
4
4
|
/**
|
|
5
5
|
* Create the MCP server.
|
|
6
6
|
* @param todoistApiKey - The API key for the todoist account.
|
package/dist/mcp-server.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mcp-server.d.ts","sourceRoot":"","sources":["../src/mcp-server.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,MAAM,yCAAyC,CAAA;AAGnE,OAAO,EACH,aAAa,EACb,KAAK,OAAO,EACZ,KAAK,WAAW,EAChB,KAAK,QAAQ,EAEhB,MAAM,kBAAkB,CAAA;AAKzB,eAAO,MAAM,YAAY,
|
|
1
|
+
{"version":3,"file":"mcp-server.d.ts","sourceRoot":"","sources":["../src/mcp-server.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,MAAM,yCAAyC,CAAA;AAGnE,OAAO,EACH,aAAa,EACb,KAAK,OAAO,EACZ,KAAK,WAAW,EAChB,KAAK,QAAQ,EAEhB,MAAM,kBAAkB,CAAA;AAKzB,eAAO,MAAM,YAAY,m1HAmDxB,CAAA;AAED;;;;;;GAMG;AACH,iBAAS,YAAY,CAAC,EAClB,aAAa,EACb,OAAO,EACP,QAAa,GAChB,EAAE;IACC,aAAa,EAAE,MAAM,CAAA;IACrB,OAAO,CAAC,EAAE,MAAM,CAAA;IAChB,QAAQ,CAAC,EAAE,QAAQ,CAAA;CACtB,aA4CA;AAED,OAAO,EAAE,aAAa,EAAE,KAAK,OAAO,EAAE,KAAK,WAAW,EAAE,KAAK,QAAQ,EAAE,YAAY,EAAE,CAAA"}
|
package/dist/tool-registry.d.ts
CHANGED
|
@@ -2891,10 +2891,6 @@ declare const toolRegistry: {
|
|
|
2891
2891
|
}>;
|
|
2892
2892
|
description: import('zod').ZodOptional<import('zod').ZodString>;
|
|
2893
2893
|
descriptionSummary: import('zod').ZodOptional<import('zod').ZodString>;
|
|
2894
|
-
taskRecommendations: import('zod').ZodOptional<import('zod').ZodArray<import('zod').ZodObject<{
|
|
2895
|
-
taskId: import('zod').ZodString;
|
|
2896
|
-
recommendation: import('zod').ZodString;
|
|
2897
|
-
}, import('zod/v4/core').$strip>>>;
|
|
2898
2894
|
isStale: import('zod').ZodBoolean;
|
|
2899
2895
|
updateInProgress: import('zod').ZodBoolean;
|
|
2900
2896
|
updatedAt: import('zod').ZodOptional<import('zod').ZodString>;
|
|
@@ -2942,10 +2938,6 @@ declare const toolRegistry: {
|
|
|
2942
2938
|
status: "UNKNOWN" | "ON_TRACK" | "AT_RISK" | "CRITICAL" | "EXCELLENT" | "ERROR";
|
|
2943
2939
|
description: string | undefined;
|
|
2944
2940
|
descriptionSummary: string | undefined;
|
|
2945
|
-
taskRecommendations: {
|
|
2946
|
-
taskId: string;
|
|
2947
|
-
recommendation: string;
|
|
2948
|
-
}[] | undefined;
|
|
2949
2941
|
isStale: boolean;
|
|
2950
2942
|
updateInProgress: boolean;
|
|
2951
2943
|
updatedAt: string | undefined;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"tool-registry.d.ts","sourceRoot":"","sources":["../src/tool-registry.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAA;AAgEvD;;;;;;;;;;GAUG;AACH,QAAA,MAAM,YAAY
|
|
1
|
+
{"version":3,"file":"tool-registry.d.ts","sourceRoot":"","sources":["../src/tool-registry.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAA;AAgEvD;;;;;;;;;;GAUG;AACH,QAAA,MAAM,YAAY;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA6EjB,CAAA;AAED;;;;;GAKG;AACH,QAAA,MAAM,eAAe,EAAE,SAAS,cAAc,EAAgC,CAAA;AAE9E,OAAO,EAAE,eAAe,EAAE,YAAY,EAAE,CAAA"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"find-activity.d.ts","sourceRoot":"","sources":["../../src/tools/find-activity.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,CAAC,EAAE,MAAM,KAAK,CAAA;
|
|
1
|
+
{"version":3,"file":"find-activity.d.ts","sourceRoot":"","sources":["../../src/tools/find-activity.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,CAAC,EAAE,MAAM,KAAK,CAAA;AAkFvB,QAAA,MAAM,YAAY;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAsD6C,CAAA;AA6I/D,OAAO,EAAE,YAAY,EAAE,CAAA"}
|
|
@@ -26,10 +26,6 @@ declare const getProjectHealth: {
|
|
|
26
26
|
}>;
|
|
27
27
|
description: z.ZodOptional<z.ZodString>;
|
|
28
28
|
descriptionSummary: z.ZodOptional<z.ZodString>;
|
|
29
|
-
taskRecommendations: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
30
|
-
taskId: z.ZodString;
|
|
31
|
-
recommendation: z.ZodString;
|
|
32
|
-
}, z.core.$strip>>>;
|
|
33
29
|
isStale: z.ZodBoolean;
|
|
34
30
|
updateInProgress: z.ZodBoolean;
|
|
35
31
|
updatedAt: z.ZodOptional<z.ZodString>;
|
|
@@ -77,10 +73,6 @@ declare const getProjectHealth: {
|
|
|
77
73
|
status: "UNKNOWN" | "ON_TRACK" | "AT_RISK" | "CRITICAL" | "EXCELLENT" | "ERROR";
|
|
78
74
|
description: string | undefined;
|
|
79
75
|
descriptionSummary: string | undefined;
|
|
80
|
-
taskRecommendations: {
|
|
81
|
-
taskId: string;
|
|
82
|
-
recommendation: string;
|
|
83
|
-
}[] | undefined;
|
|
84
76
|
isStale: boolean;
|
|
85
77
|
updateInProgress: boolean;
|
|
86
78
|
updatedAt: string | undefined;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"get-project-health.d.ts","sourceRoot":"","sources":["../../src/tools/get-project-health.ts"],"names":[],"mappings":"AAAA,OAAO,EAKH,KAAK,UAAU,EAClB,MAAM,oBAAoB,CAAA;AAC3B,OAAO,EAAE,CAAC,EAAE,MAAM,KAAK,CAAA;
|
|
1
|
+
{"version":3,"file":"get-project-health.d.ts","sourceRoot":"","sources":["../../src/tools/get-project-health.ts"],"names":[],"mappings":"AAAA,OAAO,EAKH,KAAK,UAAU,EAClB,MAAM,oBAAoB,CAAA;AAC3B,OAAO,EAAE,CAAC,EAAE,MAAM,KAAK,CAAA;AA4KvB,QAAA,MAAM,gBAAgB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA+DyC,CAAA;AAE/D,OAAO,EAAE,gBAAgB,EAAE,CAAA"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@doist/todoist-mcp",
|
|
3
|
-
"version": "12.5.
|
|
3
|
+
"version": "12.5.7",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"main": "./dist/index.js",
|
|
6
6
|
"types": "./dist/index.d.ts",
|
|
@@ -65,7 +65,7 @@
|
|
|
65
65
|
"prepare": "husky"
|
|
66
66
|
},
|
|
67
67
|
"dependencies": {
|
|
68
|
-
"@doist/todoist-sdk": "
|
|
68
|
+
"@doist/todoist-sdk": "14.0.0",
|
|
69
69
|
"@modelcontextprotocol/ext-apps": "1.2.2",
|
|
70
70
|
"date-fns": "4.1.0",
|
|
71
71
|
"dompurify": "3.3.3",
|
|
@@ -14,9 +14,11 @@
|
|
|
14
14
|
* git stash pop
|
|
15
15
|
* npx tsx scripts/eval-instructions.ts --label after
|
|
16
16
|
*
|
|
17
|
-
* Tool calls are never executed
|
|
18
|
-
*
|
|
19
|
-
*
|
|
17
|
+
* Tool calls are never executed against Todoist, so this touches no Todoist
|
|
18
|
+
* data and needs no Todoist token. Judging looks at one call per attempt: the
|
|
19
|
+
* first, unless it is a context tool (see CONTEXT_TOOLS), in which case a
|
|
20
|
+
* canned result is fed back and the next call is judged. It does spend money
|
|
21
|
+
* on model calls; see --repeats and --models.
|
|
20
22
|
*
|
|
21
23
|
* Auth uses the Anthropic SDK's standard credential chain: either
|
|
22
24
|
* ANTHROPIC_API_KEY (a key from the Anthropic Console at platform.claude.com),
|
|
@@ -34,9 +36,17 @@ import Anthropic from '@anthropic-ai/sdk'
|
|
|
34
36
|
import { z } from 'zod'
|
|
35
37
|
import { instructions } from '../src/mcp-server.js'
|
|
36
38
|
import { registeredTools } from '../src/tool-registry.js'
|
|
39
|
+
import type { UserInfoStructured } from '../src/tools/user-info.js'
|
|
37
40
|
import { createLimiter } from '../src/utils/concurrency.js'
|
|
38
41
|
import { ToolNames } from '../src/utils/tool-names.js'
|
|
39
42
|
|
|
43
|
+
/**
|
|
44
|
+
* The user ID the stubbed `user-info` reports. A scenario asking a first-person
|
|
45
|
+
* question can assert the model carried it into a filter argument rather than
|
|
46
|
+
* querying everyone.
|
|
47
|
+
*/
|
|
48
|
+
const EVAL_USER_ID = '2671355'
|
|
49
|
+
|
|
40
50
|
type Check = (input: Record<string, unknown>) => string | null
|
|
41
51
|
|
|
42
52
|
/** First item of a batch tool's array argument, e.g. add-tasks' `tasks`. */
|
|
@@ -89,9 +99,14 @@ const SCENARIOS: Scenario[] = [
|
|
|
89
99
|
if (!input.dateFrom || !input.dateTo) {
|
|
90
100
|
return 'no dateFrom/dateTo, so this searches all history rather than last week'
|
|
91
101
|
}
|
|
102
|
+
// find-activity reports every user's events by default, so "what did
|
|
103
|
+
// I get done" answered without this includes collaborators' work.
|
|
104
|
+
if (input.initiatorId !== EVAL_USER_ID) {
|
|
105
|
+
return `initiatorId is ${JSON.stringify(input.initiatorId)}, so this reports every collaborator's completions, not "I"`
|
|
106
|
+
}
|
|
92
107
|
return null
|
|
93
108
|
},
|
|
94
|
-
guards: 'instructions: find-activity vs find-completed-tasks, over a bounded range',
|
|
109
|
+
guards: 'instructions: find-activity vs find-completed-tasks, over a bounded range, filtered to the asker',
|
|
95
110
|
},
|
|
96
111
|
{
|
|
97
112
|
id: 'resolve-person',
|
|
@@ -195,6 +210,51 @@ const DEFAULT_MODELS = ['claude-haiku-4-5', 'claude-sonnet-5']
|
|
|
195
210
|
const DEFAULT_REPEATS = 5
|
|
196
211
|
const MAX_CONCURRENCY = 4
|
|
197
212
|
|
|
213
|
+
/**
|
|
214
|
+
* Tools a model may legitimately call before the one a scenario is testing, to
|
|
215
|
+
* establish context it has no other way to get — today's date, the user's
|
|
216
|
+
* timezone. "What did I get done last week?" cannot be turned into a date range
|
|
217
|
+
* without them.
|
|
218
|
+
*
|
|
219
|
+
* Judging the very first call outright scores that correct behaviour as a
|
|
220
|
+
* failure: an `expect` allowlist naming the tool under test cannot also name
|
|
221
|
+
* every reasonable lookup that precedes it. So a call to one of these is
|
|
222
|
+
* answered with the canned result below and judging moves to the next call.
|
|
223
|
+
*
|
|
224
|
+
* The result is fiction, but its shape has to be real: a scenario may assert
|
|
225
|
+
* that the model carried a value through (see EVAL_USER_ID), and a model reads
|
|
226
|
+
* a malformed field the way it would read any other bad data. `satisfies` ties
|
|
227
|
+
* it to the tool's own output type, so a renamed or added field breaks here
|
|
228
|
+
* rather than drifting quietly. Formats must match too — `currentLocalTime`
|
|
229
|
+
* is what `toLocaleString('en-US', …)` produces, not ISO.
|
|
230
|
+
*/
|
|
231
|
+
const CONTEXT_TOOLS: Record<string, unknown> = {
|
|
232
|
+
[ToolNames.USER_INFO]: {
|
|
233
|
+
type: 'user_info',
|
|
234
|
+
userId: EVAL_USER_ID,
|
|
235
|
+
fullName: 'Eval User',
|
|
236
|
+
timezone: 'Europe/London',
|
|
237
|
+
currentLocalTime: '08/11/2026, 09:30:00',
|
|
238
|
+
startDay: 1,
|
|
239
|
+
startDayName: 'Monday',
|
|
240
|
+
weekStartDate: '2026-08-10',
|
|
241
|
+
weekEndDate: '2026-08-16',
|
|
242
|
+
currentWeekNumber: 33,
|
|
243
|
+
completedToday: 3,
|
|
244
|
+
dailyGoal: 5,
|
|
245
|
+
weeklyGoal: 30,
|
|
246
|
+
email: 'eval@example.com',
|
|
247
|
+
plan: 'Todoist Pro',
|
|
248
|
+
} satisfies UserInfoStructured,
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
/**
|
|
252
|
+
* How many context calls an attempt may make before it is judged a failure.
|
|
253
|
+
* A model that keeps gathering context is not answering the question, and
|
|
254
|
+
* without a cap a loop would bill for turns forever.
|
|
255
|
+
*/
|
|
256
|
+
const MAX_CONTEXT_HOPS = 3
|
|
257
|
+
|
|
198
258
|
function parseArgs() {
|
|
199
259
|
const args = process.argv.slice(2)
|
|
200
260
|
const get = (flag: string) => {
|
|
@@ -254,6 +314,28 @@ type Attempt = {
|
|
|
254
314
|
*/
|
|
255
315
|
errored: boolean
|
|
256
316
|
usage: Usage
|
|
317
|
+
/** Context calls answered with a stub before the judged one. */
|
|
318
|
+
contextHops: number
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
/** Verdict on the one call a scenario is judged by. */
|
|
322
|
+
function judge(
|
|
323
|
+
scenario: Scenario,
|
|
324
|
+
call: Anthropic.ToolUseBlock,
|
|
325
|
+
): { pass: boolean; reason: string } {
|
|
326
|
+
if (scenario.forbid?.includes(call.name)) {
|
|
327
|
+
return { pass: false, reason: `called ${call.name}, which this rule forbids` }
|
|
328
|
+
}
|
|
329
|
+
if (scenario.expect && !scenario.expect.includes(call.name)) {
|
|
330
|
+
return { pass: false, reason: `expected ${scenario.expect.join(' or ')}` }
|
|
331
|
+
}
|
|
332
|
+
// An argument check written for a specific tool must not run against a
|
|
333
|
+
// different one a forbid-only scenario legitimately allows.
|
|
334
|
+
if (scenario.forbid && !scenario.expect) {
|
|
335
|
+
return { pass: true, reason: '' }
|
|
336
|
+
}
|
|
337
|
+
const reason = scenario.check?.(call.input as Record<string, unknown>) ?? null
|
|
338
|
+
return { pass: reason === null, reason: reason ?? '' }
|
|
257
339
|
}
|
|
258
340
|
|
|
259
341
|
async function runAttempt(
|
|
@@ -267,51 +349,77 @@ async function runAttempt(
|
|
|
267
349
|
model,
|
|
268
350
|
errored: false,
|
|
269
351
|
usage: { input: 0, output: 0, cacheWrite: 0, cacheRead: 0 },
|
|
352
|
+
contextHops: 0,
|
|
270
353
|
}
|
|
354
|
+
const messages: Anthropic.MessageParam[] = [{ role: 'user', content: scenario.prompt }]
|
|
271
355
|
try {
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
356
|
+
for (let hop = 0; ; hop++) {
|
|
357
|
+
const response = await client.messages.create({
|
|
358
|
+
model,
|
|
359
|
+
max_tokens: 4096,
|
|
360
|
+
// Tools render before system, so one breakpoint here caches both.
|
|
361
|
+
system: [
|
|
362
|
+
{ type: 'text', text: instructions, cache_control: { type: 'ephemeral' } },
|
|
363
|
+
],
|
|
364
|
+
tools,
|
|
365
|
+
messages,
|
|
366
|
+
})
|
|
280
367
|
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
368
|
+
base.usage = {
|
|
369
|
+
input: base.usage.input + response.usage.input_tokens,
|
|
370
|
+
output: base.usage.output + response.usage.output_tokens,
|
|
371
|
+
cacheWrite:
|
|
372
|
+
base.usage.cacheWrite + (response.usage.cache_creation_input_tokens ?? 0),
|
|
373
|
+
cacheRead: base.usage.cacheRead + (response.usage.cache_read_input_tokens ?? 0),
|
|
374
|
+
}
|
|
287
375
|
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
if (scenario.forbid?.includes(call.name)) {
|
|
293
|
-
return {
|
|
294
|
-
...base,
|
|
295
|
-
calledTool: call.name,
|
|
296
|
-
pass: false,
|
|
297
|
-
reason: `called ${call.name}, which this rule forbids`,
|
|
376
|
+
const calls = response.content.filter((b) => b.type === 'tool_use')
|
|
377
|
+
const call = calls[0]
|
|
378
|
+
if (!call) {
|
|
379
|
+
return { ...base, calledTool: null, pass: false, reason: 'no tool call' }
|
|
298
380
|
}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
381
|
+
|
|
382
|
+
// A turn can carry several calls at once. Judge the substantive one
|
|
383
|
+
// rather than whichever block came first: a model that asks for the
|
|
384
|
+
// date and queries the activity log in the same turn has made its
|
|
385
|
+
// choice, and the order between the two is arbitrary. A forbidden
|
|
386
|
+
// call outranks that — catching it is the point of a forbid rule.
|
|
387
|
+
//
|
|
388
|
+
// A tool a scenario expects is never treated as context, or a
|
|
389
|
+
// scenario testing the route *to* user-info could never pass: its
|
|
390
|
+
// expected call would be stubbed instead of judged.
|
|
391
|
+
const isContext = (c: Anthropic.ToolUseBlock) =>
|
|
392
|
+
Object.hasOwn(CONTEXT_TOOLS, c.name) && !scenario.expect?.includes(c.name)
|
|
393
|
+
const substantive =
|
|
394
|
+
calls.find((c) => scenario.forbid?.includes(c.name)) ??
|
|
395
|
+
calls.find((c) => !isContext(c))
|
|
396
|
+
if (substantive) {
|
|
397
|
+
const { pass, reason } = judge(scenario, substantive)
|
|
398
|
+
return { ...base, calledTool: substantive.name, pass, reason: reason || null }
|
|
306
399
|
}
|
|
400
|
+
|
|
401
|
+
// Nothing but context gathering, so answer it and judge what the
|
|
402
|
+
// model reaches for next.
|
|
403
|
+
if (hop >= MAX_CONTEXT_HOPS) {
|
|
404
|
+
return {
|
|
405
|
+
...base,
|
|
406
|
+
calledTool: call.name,
|
|
407
|
+
pass: false,
|
|
408
|
+
reason: `still gathering context after ${MAX_CONTEXT_HOPS} calls`,
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
base.contextHops = hop + 1
|
|
413
|
+
messages.push({ role: 'assistant', content: response.content })
|
|
414
|
+
messages.push({
|
|
415
|
+
role: 'user',
|
|
416
|
+
content: calls.map((c) => ({
|
|
417
|
+
type: 'tool_result' as const,
|
|
418
|
+
tool_use_id: c.id,
|
|
419
|
+
content: JSON.stringify(CONTEXT_TOOLS[c.name]),
|
|
420
|
+
})),
|
|
421
|
+
})
|
|
307
422
|
}
|
|
308
|
-
// An argument check written for a specific tool must not run against a
|
|
309
|
-
// different one a forbid-only scenario legitimately allows.
|
|
310
|
-
if (scenario.forbid && !scenario.expect) {
|
|
311
|
-
return { ...base, calledTool: call.name, pass: true, reason: null }
|
|
312
|
-
}
|
|
313
|
-
const reason = scenario.check?.(call.input as Record<string, unknown>) ?? null
|
|
314
|
-
return { ...base, calledTool: call.name, pass: reason === null, reason }
|
|
315
423
|
} catch (error) {
|
|
316
424
|
const message = error instanceof Error ? error.message : String(error)
|
|
317
425
|
return {
|