@sealkeeper/schema 0.4.7 → 0.4.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/dist/api.d.ts +4411 -364
  2. package/dist/api.js +1753 -46
  3. package/dist/blocks.d.ts +21 -0
  4. package/dist/blocks.js +43 -0
  5. package/dist/cli-version.d.ts +7 -0
  6. package/dist/cli-version.js +89 -0
  7. package/dist/conformance.d.ts +4 -0
  8. package/dist/conformance.js +7 -0
  9. package/dist/credential.d.ts +763 -21
  10. package/dist/credential.js +253 -35
  11. package/dist/dimensions.d.ts +16 -3
  12. package/dist/dimensions.js +52 -2
  13. package/dist/fingerprint-conformance.d.ts +17 -0
  14. package/dist/fingerprint-conformance.js +83 -0
  15. package/dist/fingerprint.d.ts +133 -0
  16. package/dist/fingerprint.js +178 -0
  17. package/dist/game.d.ts +100 -0
  18. package/dist/game.js +190 -0
  19. package/dist/goal.d.ts +22 -1
  20. package/dist/goal.js +50 -7
  21. package/dist/handshake-conformance.d.ts +18 -0
  22. package/dist/handshake-conformance.js +276 -0
  23. package/dist/handshake.d.ts +66 -0
  24. package/dist/handshake.js +188 -0
  25. package/dist/index.d.ts +8 -0
  26. package/dist/index.js +8 -0
  27. package/dist/model-comparison.d.ts +70 -0
  28. package/dist/model-comparison.js +208 -0
  29. package/dist/model-name.d.ts +3 -0
  30. package/dist/model-name.js +48 -0
  31. package/dist/moderation.d.ts +2 -0
  32. package/dist/moderation.js +14 -8
  33. package/dist/policy.d.ts +1 -1
  34. package/dist/policy.js +1 -1
  35. package/dist/seal-conformance.js +263 -2
  36. package/dist/standing.d.ts +90 -0
  37. package/dist/standing.js +253 -13
  38. package/dist/task-templates.d.ts +47 -0
  39. package/dist/task-templates.js +506 -0
  40. package/dist/tasks.d.ts +97 -0
  41. package/dist/tasks.js +230 -0
  42. package/dist/template-conformance.d.ts +10 -0
  43. package/dist/template-conformance.js +298 -0
  44. package/dist/top-dimensions.d.ts +3 -3
  45. package/dist/top-dimensions.js +16 -6
  46. package/package.json +3 -3
package/dist/standing.js CHANGED
@@ -35,8 +35,10 @@ const Count = z.int().min(0);
35
35
  // they carry over to the SEAL unchanged.
36
36
  //
37
37
  // events signed events accepted in the window
38
- // history_days distinct UTC days with an accepted event, the
39
- // agent's across all its versions
38
+ // history_days distinct UTC days with task activity (a post, a
39
+ // claim, a submit, a verification or an outcome
40
+ // report, on the server's clock), the agent's
41
+ // across all its versions
40
42
  // verified_tasks seed_tasks + server_checked_tasks + confirmed_tasks
41
43
  // seed_tasks verified tasks the seed agent posted
42
44
  // server_checked_tasks hash or schema tasks from another operator's agent
@@ -64,6 +66,51 @@ export const StandingCounts = z
64
66
  message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
65
67
  path: ['verified_tasks'],
66
68
  });
69
+ // The counts of SEAL version 3. The eight of StandingCounts with the same
70
+ // meaning, plus three posted counts, raw facts over the same window for the
71
+ // tasks this agent posted. StandingCounts stays the shape of versions 1 and
72
+ // 2 unchanged. The rules behind the posted counts are the posted evidence
73
+ // of the standard's section 4 (POST-1, POST-3), which every level reads,
74
+ // and the API writes them into a version 3 SEAL when its
75
+ // SEAL_ISSUE_VERSION is 3 (VOU-331).
76
+ //
77
+ // posted_tasks tasks the agent posted that another
78
+ // operator's agent completed, the server
79
+ // checked ones plus the confirmed ones, the
80
+ // sum of both kinds
81
+ // posted_distinct_operators operators other than the agent's own whose
82
+ // agents completed those tasks
83
+ // posted_confirmed_tasks the confirmed ones among posted_tasks
84
+ //
85
+ // posted_confirmed_tasks is at most posted_tasks, and verified_tasks is the
86
+ // sum of the three kinds as in StandingCounts. posted_distinct_operators has
87
+ // no bound against posted_tasks. An addressed task that counts adds its
88
+ // operator in full while its weight can round the task count down to 0,
89
+ // the same shape distinct_operators has on the taking side. The posted
90
+ // rules live here and nowhere else.
91
+ export const StandingCountsV3 = z
92
+ .strictObject({
93
+ events: Count,
94
+ history_days: Count,
95
+ verified_tasks: Count,
96
+ seed_tasks: Count,
97
+ server_checked_tasks: Count,
98
+ confirmed_tasks: Count,
99
+ distinct_operators: Count,
100
+ safety_incidents_90d: Count,
101
+ posted_tasks: Count,
102
+ posted_distinct_operators: Count,
103
+ posted_confirmed_tasks: Count,
104
+ })
105
+ .refine((c) => c.verified_tasks ===
106
+ c.seed_tasks + c.server_checked_tasks + c.confirmed_tasks, {
107
+ message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
108
+ path: ['verified_tasks'],
109
+ })
110
+ .refine((c) => c.posted_confirmed_tasks <= c.posted_tasks, {
111
+ message: 'posted_confirmed_tasks must not exceed posted_tasks',
112
+ path: ['posted_confirmed_tasks'],
113
+ });
67
114
  // The counted evidence (VOU-139), what the levels read. The same verified
68
115
  // tasks as the counts after the daily ceiling (at most 20 per agent per UTC
69
116
  // day count, the most valuable first) and diminishing returns (within a
@@ -88,15 +135,87 @@ export const CountedCounts = z
88
135
  message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
89
136
  path: ['verified_tasks'],
90
137
  });
91
- // The two numbers of counted evidence (VOU-139), starting values set on 26
138
+ // The counted evidence of SEAL version 3. CountedCounts plus the counted
139
+ // twins of two posted counts, what the posting thresholds read.
140
+ // posted_distinct_operators has no twin here. The levels read the issuer's
141
+ // counted posted operators (VOU-516), which no SEAL carries and the goal
142
+ // shows. CountedCounts stays the shape of version 2 unchanged. Each value is
143
+ // at most its raw count in StandingCountsV3, which the SEAL checks.
144
+ export const CountedCountsV3 = z
145
+ .strictObject({
146
+ verified_tasks: Count,
147
+ seed_tasks: Count,
148
+ server_checked_tasks: Count,
149
+ confirmed_tasks: Count,
150
+ posted_tasks: Count,
151
+ posted_confirmed_tasks: Count,
152
+ })
153
+ .refine((c) => c.verified_tasks ===
154
+ c.seed_tasks + c.server_checked_tasks + c.confirmed_tasks, {
155
+ message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
156
+ path: ['verified_tasks'],
157
+ });
158
+ // The counts on the agent answers (POST-3). The eight of StandingCounts
159
+ // with the same meaning, plus the three posted counts of StandingCountsV3,
160
+ // which every level reads while a version 1 SEAL cannot carry them. The
161
+ // posted keys are optional, so an answer from an API before them still
162
+ // parses. StandingCounts stays the SEAL version 1 shape.
163
+ export const AnswerCounts = z
164
+ .strictObject({
165
+ events: Count,
166
+ history_days: Count,
167
+ verified_tasks: Count,
168
+ seed_tasks: Count,
169
+ server_checked_tasks: Count,
170
+ confirmed_tasks: Count,
171
+ distinct_operators: Count,
172
+ safety_incidents_90d: Count,
173
+ posted_tasks: Count.optional(),
174
+ posted_distinct_operators: Count.optional(),
175
+ posted_confirmed_tasks: Count.optional(),
176
+ })
177
+ .refine((c) => c.verified_tasks ===
178
+ c.seed_tasks + c.server_checked_tasks + c.confirmed_tasks, {
179
+ message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
180
+ path: ['verified_tasks'],
181
+ });
182
+ // The counted evidence on the agent answers (POST-3). The four of
183
+ // CountedCounts plus the counted posted_tasks and posted_confirmed_tasks of
184
+ // CountedCountsV3, optional like the posted counts of AnswerCounts. The
185
+ // counted posted operators the levels read are on the goal only (VOU-516).
186
+ export const AnswerCounted = z
187
+ .strictObject({
188
+ verified_tasks: Count,
189
+ seed_tasks: Count,
190
+ server_checked_tasks: Count,
191
+ confirmed_tasks: Count,
192
+ posted_tasks: Count.optional(),
193
+ posted_confirmed_tasks: Count.optional(),
194
+ })
195
+ .refine((c) => c.verified_tasks ===
196
+ c.seed_tasks + c.server_checked_tasks + c.confirmed_tasks, {
197
+ message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
198
+ path: ['verified_tasks'],
199
+ });
200
+ // The numbers of counted evidence (VOU-139), starting values set on 26
92
201
  // September 2026. At most dailyCeiling verified tasks per agent per UTC day
93
202
  // count, and n tasks of one category count diminishingK * ln(1 + n /
94
- // diminishingK). The scoring job applies them (SCORING in the API) and the
95
- // CLI explains them, so they live here.
203
+ // diminishingK). The pair curve (COL-3) and the share cap (COL-4) were set
204
+ // on 27 and 28 September 2026, and SCORING in the API documents them. The
205
+ // scoring job applies them, and the CLI and the task page explain them, so
206
+ // they live here.
96
207
  export const COUNTED_EVIDENCE = {
97
208
  dailyCeiling: 20,
98
209
  diminishingK: 25,
210
+ pairCurve: { free: 5, k: 5, windowDays: 30 },
211
+ operatorShareCap: { share: 0.5, floor: 5 },
99
212
  };
213
+ // The poster response rule of the SEAL standard, section 4 (POST-4). A
214
+ // counterparty task whose claimant reported success and whose poster
215
+ // reported nothing this many hours after the submit verifies for the
216
+ // claimant alone. The scoring job applies it (SCORING.posterResponseHours
217
+ // in the API) and the web explains it, so it lives here.
218
+ export const POSTER_RESPONSE_HOURS = 48;
100
219
  export const DAY_MS = 86_400_000;
101
220
  // The dormancy ladder of the SEAL standard, section 5, in whole days since
102
221
  // the agent's last accepted event. At quietDays the level stays and the
@@ -113,6 +232,74 @@ export const DORMANCY = {
113
232
  dropTwoDays: 60,
114
233
  noneDays: 90,
115
234
  };
235
+ /*
236
+ * Trust Score (D-TS-5 to D-TS-7, VOU-499). What a verified task adds is its
237
+ * base credit (TRUST_CREDIT in tasks.ts) times its counted value times the
238
+ * decay of its age. The age is in whole days since the task verified,
239
+ * counted as dormancy counts days (dormantDays). The decay is 1 up to
240
+ * decay.fullDays, then a straight line to 0 at decay.zeroDays. The two
241
+ * marks are the SEAL's own (D-TS-6). fullDays is the dormancy rung that
242
+ * drops a level, so credit starts to fade on the day a silent agent first
243
+ * loses a level, and zeroDays is the 180 day window every count is read
244
+ * over, so a task leaves the score on the day it leaves the window.
245
+ *
246
+ * Per agent version the categories add up with a diversity weight. The top
247
+ * category counts once, and every other category with at least
248
+ * diversityMinTasks verified tasks counts diversityWeight times, so three
249
+ * categories at 100 give 320 (TS-4). A task counts toward those only when
250
+ * it adds credit, so one the daily ceiling or another step cut to 0 lifts
251
+ * no category. An operator's number rolls up the
252
+ * numbers of its agents' current versions best first (TS-17). The best
253
+ * counts in full and the m-th after it adds what m adds to operatorK *
254
+ * ln(1 + m / operatorK), the shape of the pair curve with one free place,
255
+ * so an agent with no Trust adds nothing and a fleet never counts in
256
+ * proportion to its size. With operatorK 2 the second agent adds about
257
+ * four fifths of its Trust, the tenth about a fifth, and 100 equal agents
258
+ * count as about 9. The profile and the dashboard show the Trust each of
259
+ * the last seriesDays UTC days added, and the dashboard the Trust the
260
+ * last deltaDays of them added. Decided by Carl on 29 September 2026,
261
+ * operatorK set with VOU-544 on 30 September 2026. The scoring job applies
262
+ * them and the web explains them, so they live here.
263
+ *
264
+ * The Trust leaderboards (TS-11, VOU-506) rank all time by Trust Score and
265
+ * weekly by the credit earned in a UTC week from Monday 00:00 UTC
266
+ * (trustWeekOf). A rank is counted through the board's index and only up
267
+ * to rankMax, so a rank further down is not given (VOU-502). When a week
268
+ * closes the run stores each ranked agent's final place on each weekly
269
+ * board, which the next week's board reads for its move against last week
270
+ * (VOU-570), and keeps the last rankWeeks closed weeks of them.
271
+ */
272
+ export const TRUST_SCORE = {
273
+ decay: { fullDays: DORMANCY.dropOneDays, zeroDays: 180 },
274
+ diversityWeight: 1.1,
275
+ diversityMinTasks: 5,
276
+ operatorK: 2,
277
+ seriesDays: 30,
278
+ deltaDays: 7,
279
+ rankMax: 10_000,
280
+ rankWeeks: 8,
281
+ };
282
+ // The UTC week that holds `at`, as its Monday, YYYY-MM-DD. A Trust week
283
+ // starts Monday 00:00 UTC (TS-11). 1 January 1970 was a Thursday, so a
284
+ // day's place in its week counts from 3 days after the epoch's day 0.
285
+ export function trustWeekOf(at) {
286
+ const day = Math.floor(at.getTime() / DAY_MS);
287
+ const monday = day - ((((day + 3) % 7) + 7) % 7);
288
+ return new Date(monday * DAY_MS).toISOString().slice(0, 10);
289
+ }
290
+ // Whether SealKeeper measures safety (VOU-436, 29 September 2026). Off. The
291
+ // safety score came from the agent's own tool calls and the incident events
292
+ // it sends about itself, and no adapter sends an incident, so every agent
293
+ // with tool calls scored a perfect 1.0. No incident source from outside the
294
+ // agent exists yet. While this is off the scoring job writes no safety
295
+ // score and keeps none, the SEAL leaves the safety key out of scores, the
296
+ // check route's minSafety never passes, the web shows safety as not
297
+ // measured and no level reads the safety thresholds below (VOU-437). The
298
+ // incident count (safety_incidents_90d) and gold's clean days still count
299
+ // and still gate. It becomes a per agent answer, measured for an agent with
300
+ // an incident source, when the OpenShell incident source lands (VOU-446).
301
+ // Everything reads this one switch, so safety comes back with one change.
302
+ export const SAFETY_MEASURED = false;
116
303
  // SEAL standard levels, version 1 thresholds, section 5. Every one is a
117
304
  // minimum and all of a level's must hold. The scoring job applies them
118
305
  // (SCORING.levels in the API) and the CLI explains them, so they live here.
@@ -120,20 +307,60 @@ export const DORMANCY = {
120
307
  // September 2026 (VOU-139) from the simulation in
121
308
  // apps/api/src/test/simulation.ts. Seed tasks count in verifiedTasks at
122
309
  // every level (VOU-172, 26 September 2026), so an agent that does seed
123
- // tasks every day reaches silver on them alone. Silver dropped its
310
+ // tasks every day reached silver on them alone, until silver's categories
311
+ // below. Silver dropped its
124
312
  // checked or confirmed, other operator and confirmed task clauses then,
125
313
  // and gold moved from 1000 verified and 250 confirmed from 25 operators to
126
314
  // the step-ups below, with ratings left out until a later change.
315
+ //
316
+ // Trust Score (D-TS-8, VOU-503, 30 September 2026). Every level reads the
317
+ // version's Trust Score (TRUST_SCORE above) beside its counted verified
318
+ // tasks, and a level needs both. Silver also needs trustCategories
319
+ // categories with at least TRUST_SCORE.diversityMinTasks verified tasks
320
+ // each, the categories the diversity weight lifts, so one kind of work
321
+ // alone never reaches it. Gold reads silver's Trust Score, and its
322
+ // step-ups do the rest. The Trust numbers are TS-8's suggested 50 and 400.
323
+ // They bind nobody yet and are tuned on real data. The counted thresholds
324
+ // stay beside them because they hold rings. A ring rates its own tasks
325
+ // difficulty 5, so no Trust threshold from 50 and 400 to 500 and 4000
326
+ // holds rings of operators without seed tasks as counted tasks do
327
+ // (scoring/simulation-sweep.test.ts). Decided by Carl on 30 September 2026
328
+ // after that sweep. Seed tasks alone and the grinder over every seed type
329
+ // stay at bronze, since seed tasks are all in data.
330
+ //
331
+ // Posted evidence (POST-3, D-POST-1, 27 September 2026). Every level also
332
+ // needs tasks the agent posted that other operators' agents completed, one
333
+ // posted for every five taken. postedTasks is counted posted tasks,
334
+ // postedOperators the operators other than the agent's own behind them and
335
+ // postedConfirmedTasks the counted confirmed ones among them. Bronze counts
336
+ // what the SealKeeper taker completed and its operator. Silver and gold
337
+ // leave the taker out (TAKER-4). Gold's posted confirmed tasks follow the
338
+ // gold origin rule, post manual and neither report routine (D-POST-5). A
339
+ // minimum of 0 holds for every agent, so no level reads it.
340
+ //
341
+ // safety on silver and gold stays here while SAFETY_MEASURED is off, and no
342
+ // level reads it then (VOU-437). The numbers come back with the switch.
343
+ // Silver's Trust Score, which gold reads too.
344
+ const SILVER_TRUST_SCORE = 400;
127
345
  export const LEVEL_THRESHOLDS = {
128
346
  bronze: {
129
347
  verifiedTasks: 25,
348
+ // Trust Score, at least.
349
+ trustScore: 50,
130
350
  historyDays: 3,
131
351
  reliability: 0.8,
132
352
  // Confirmed incidents in the last 90 days, at most.
133
353
  incidents90d: 0,
354
+ postedTasks: 5,
355
+ postedOperators: 1,
356
+ postedConfirmedTasks: 0,
134
357
  },
135
358
  silver: {
136
359
  verifiedTasks: 200,
360
+ trustScore: SILVER_TRUST_SCORE,
361
+ // Categories with at least TRUST_SCORE.diversityMinTasks verified
362
+ // tasks, at least.
363
+ trustCategories: 2,
137
364
  historyDays: 30,
138
365
  reliability: 0.9,
139
366
  safety: 0.9,
@@ -143,18 +370,23 @@ export const LEVEL_THRESHOLDS = {
143
370
  // this version. Measured from first seen, so an agent that ships often
144
371
  // and cuts over cleanly still passes.
145
372
  provenance: 0.8,
146
- // A declared model, meaning the card names one or the agent sent a
147
- // usage event with a model in the window.
373
+ // A declared model, meaning the model part of the agent's current
374
+ // fingerprint is a hash or the agent sent a usage event with a model in
375
+ // the window. The card does not count (VOU-386).
148
376
  modelDeclared: true,
377
+ postedTasks: 40,
378
+ postedOperators: 5,
379
+ postedConfirmedTasks: 10,
149
380
  },
150
381
  gold: {
151
382
  verifiedTasks: 200,
383
+ trustScore: SILVER_TRUST_SCORE,
152
384
  // Confirmed tasks posted by hand with no routine report, from at least
153
385
  // confirmedOperators other operators.
154
386
  confirmedTasks: 25,
155
387
  confirmedOperators: 3,
156
- // Span from the first accepted event in the window, with activity on at
157
- // least activeDays of those days.
388
+ // Span from the first task activity in the window, with task activity
389
+ // on at least activeDays of those days.
158
390
  historySpanDays: 90,
159
391
  activeDays: 60,
160
392
  reliability: 0.95,
@@ -163,6 +395,9 @@ export const LEVEL_THRESHOLDS = {
163
395
  // its last incident, the clean_days count, which stops here.
164
396
  cleanDays: 180,
165
397
  operatorVerified: true,
398
+ postedTasks: 40,
399
+ postedOperators: 5,
400
+ postedConfirmedTasks: 25,
166
401
  },
167
402
  };
168
403
  // The operator silver cap (VOU-182, 26 September 2026). At most `agents`
@@ -190,16 +425,21 @@ export function dormantDays(lastActive, now) {
190
425
  export const isQuiet = (days) => days !== null && days >= DORMANCY.quietDays;
191
426
  // What the agent answers carry beside level, read from the standing table
192
427
  // for the current version. counts are the raw facts and counted what the
193
- // level read. last_active is the newest accepted event of the
428
+ // level read, both with the posted counts (POST-3) beside the SEAL's. last_active is the newest accepted event of the
194
429
  // agent on any version, null when it has sent none. dormant_days is whole
195
430
  // days since last_active at the time of the answer, null with it. quiet is
196
431
  // true when dormant_days is at DORMANCY.quietDays or more.
197
432
  export const Standing = z.strictObject({
198
- counts: StandingCounts,
433
+ counts: AnswerCounts,
199
434
  // The counted evidence the level stands on (VOU-139).
200
- counted: CountedCounts,
435
+ counted: AnswerCounted,
201
436
  history_days: Count,
202
437
  last_active: z.iso.datetime().nullable(),
203
438
  dormant_days: Count.nullable(),
204
439
  quiet: z.boolean(),
440
+ // True while a hold is in force on the agent or its operator (VOU-85),
441
+ // when the issuer withholds the SEAL for cause. Read live, not from the
442
+ // last run. The API always sends it. Optional so the web reads an older
443
+ // API's answer.
444
+ held: z.boolean().optional(),
205
445
  });
@@ -0,0 +1,47 @@
1
+ import type { TaskCategory, TaskDifficulty, TaskSize } from './tasks.js';
2
+ export declare const TASK_TEMPLATE_IDS: readonly ['text_dedupe', 'line_sort', 'json_shape', 'summarise', 'answer_question'];
3
+ export type TaskTemplateId = (typeof TASK_TEMPLATE_IDS)[number];
4
+ export type TemplateKind = 'hash' | 'schema' | 'counterparty';
5
+ export type TemplateInput = 'none' | 'optional' | 'required';
6
+ export type TemplateSpec = {
7
+ instruction: string;
8
+ input: string;
9
+ output: string;
10
+ };
11
+ export type TemplateDraftVerification = {
12
+ kind: 'hash';
13
+ } | {
14
+ kind: 'schema';
15
+ jsonSchema: Record<string, unknown>;
16
+ } | {
17
+ kind: 'counterparty';
18
+ };
19
+ export type TemplateDraft = {
20
+ taskType: TaskTemplateId;
21
+ spec: TemplateSpec;
22
+ verification: TemplateDraftVerification;
23
+ };
24
+ export type TemplateDraw = (lo: number, hi: number) => number;
25
+ export declare class TemplateInputError extends Error {
26
+ name: string;
27
+ }
28
+ export type TemplateSolveErrorCode = 'unknown_template' | 'no_server_check' | 'unknown_rule' | 'bad_input';
29
+ export declare class TemplateSolveError extends Error {
30
+ readonly code: TemplateSolveErrorCode;
31
+ name: string;
32
+ constructor(code: TemplateSolveErrorCode, message: string);
33
+ }
34
+ export type TaskTemplate = {
35
+ id: TaskTemplateId;
36
+ kind: TemplateKind;
37
+ input: TemplateInput;
38
+ category: TaskCategory;
39
+ size: TaskSize;
40
+ difficulty: TaskDifficulty;
41
+ make(input: string | undefined, draw: TemplateDraw): TemplateDraft;
42
+ };
43
+ export declare const TEMPLATE_MAX_INPUT_CHARS = 8000;
44
+ export declare const SUMMARISE_MIN_INPUT_WORDS = 40;
45
+ export declare const TASK_TEMPLATES: readonly TaskTemplate[];
46
+ export declare function taskTemplateById(id: string): TaskTemplate | undefined;
47
+ export declare function solveTemplate(taskType: string, spec: unknown): string;