@sealkeeper/schema 0.4.8 → 0.4.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api.d.ts +4178 -906
- package/dist/api.js +1438 -40
- package/dist/blocks.d.ts +21 -0
- package/dist/blocks.js +43 -0
- package/dist/cli-version.d.ts +7 -0
- package/dist/cli-version.js +89 -0
- package/dist/credential.d.ts +518 -10
- package/dist/credential.js +160 -28
- package/dist/dimensions.d.ts +6 -6
- package/dist/dimensions.js +11 -7
- package/dist/fingerprint.d.ts +1 -0
- package/dist/fingerprint.js +12 -0
- package/dist/game.d.ts +100 -0
- package/dist/game.js +190 -0
- package/dist/goal.d.ts +7 -1
- package/dist/goal.js +26 -6
- package/dist/handshake.js +5 -4
- package/dist/index.d.ts +5 -0
- package/dist/index.js +5 -0
- package/dist/model-comparison.d.ts +70 -0
- package/dist/model-comparison.js +208 -0
- package/dist/model-name.d.ts +3 -0
- package/dist/model-name.js +48 -0
- package/dist/moderation.d.ts +2 -0
- package/dist/moderation.js +14 -8
- package/dist/policy.d.ts +1 -1
- package/dist/policy.js +1 -1
- package/dist/seal-conformance.js +120 -3
- package/dist/standing.d.ts +29 -0
- package/dist/standing.js +126 -14
- package/dist/task-templates.d.ts +2 -1
- package/dist/task-templates.js +6 -1
- package/dist/tasks.d.ts +57 -3
- package/dist/tasks.js +135 -8
- package/dist/top-dimensions.js +7 -3
- package/package.json +1 -1
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import { EventPayload } from './events.js';
|
|
2
|
+
import { foldLetters, OWN_NAMES } from './moderation.js';
|
|
3
|
+
/*
|
|
4
|
+
* The name of the model an agent runs, as the agent declares it (VOU-566).
|
|
5
|
+
* It travels as text beside the fingerprint on a sync, see
|
|
6
|
+
* FingerprintDeclaration, so SealKeeper can show and compare the models
|
|
7
|
+
* agents run across Claude, GPT, Gemini and the rest. It is what the agent
|
|
8
|
+
* says about itself, inside a signed report. Nothing reads it as proof, and
|
|
9
|
+
* it earns no Trust. The model part of the fingerprint stays a hash.
|
|
10
|
+
*/
|
|
11
|
+
const OWN_KEYS = OWN_NAMES.map(foldLetters);
|
|
12
|
+
// The usage event's model name, the same characters and the same 64 cap, so
|
|
13
|
+
// a name an adapter already sends on usage is a name it can declare. A name
|
|
14
|
+
// must also leave a key, so one that ends in a slash, or is only a date, is
|
|
15
|
+
// refused. It shows on public pages, so the moderation rules names have
|
|
16
|
+
// apply as far as they fit a model name. It may not hold SealKeeper's own
|
|
17
|
+
// names anywhere in its letters and digits as foldLetters folds them, so
|
|
18
|
+
// sealkeeper-verified, official-vouched and 5ea1keeper are refused, nor in
|
|
19
|
+
// its key's, which GET /v1/models shows (VOU-551). foldLetters leaves the
|
|
20
|
+
// rn and vv pairs alone, since normaliseName folds v/vouched to wouched
|
|
21
|
+
// while its key is vouched, and sealkeepernova to sealkeepemova. The
|
|
22
|
+
// other protected names stay allowed, since claude-opus-4-5 and gpt-4.1
|
|
23
|
+
// name their makers' models, and the characters are Latin alone, so no
|
|
24
|
+
// lookalike of another script gets in.
|
|
25
|
+
export const ModelName = EventPayload.usage.shape.model
|
|
26
|
+
.unwrap()
|
|
27
|
+
.refine((name) => modelKey(name) !== '', 'A model name needs a name after its last slash')
|
|
28
|
+
.refine((name) => ![foldLetters(name), foldLetters(modelKey(name))].some((form) => OWN_KEYS.some((own) => form.includes(own))), 'A model name may not name SealKeeper');
|
|
29
|
+
// When two names are the same model. Lower case, everything up to the last
|
|
30
|
+
// slash dropped, so a provider or router prefix goes, a trailing date of 8
|
|
31
|
+
// digits dropped with its dash, and dots made dashes, so claude-sonnet-4.5
|
|
32
|
+
// and claude-sonnet-4-5 meet. A date written with dashes, as in
|
|
33
|
+
// gpt-4.1-2025-04-14, stays, and so does anything else the rule does not
|
|
34
|
+
// name. The one function that decides it, so the API, the CLI and the web
|
|
35
|
+
// agree.
|
|
36
|
+
export function modelKey(name) {
|
|
37
|
+
const lower = name.toLowerCase();
|
|
38
|
+
return lower
|
|
39
|
+
.slice(lower.lastIndexOf('/') + 1)
|
|
40
|
+
.replace(/-\d{8}$/, '')
|
|
41
|
+
.replaceAll('.', '-');
|
|
42
|
+
}
|
|
43
|
+
// The most model changes one agent records in any 24 hours. A real change
|
|
44
|
+
// of model is rare, and an operator who tries a model and goes back makes
|
|
45
|
+
// two in a day, so 3 leaves one to spare. Past it the name is still kept
|
|
46
|
+
// as the agent's current one and no change is recorded, so an agent cannot
|
|
47
|
+
// grow the table by flipping its name.
|
|
48
|
+
export const MODEL_CHANGES_PER_DAY = 3;
|
package/dist/moderation.d.ts
CHANGED
|
@@ -8,7 +8,9 @@ export type OperatorSlug = z.infer<typeof OperatorSlug>;
|
|
|
8
8
|
export declare const OperatorDisplayName: z.ZodString;
|
|
9
9
|
export type OperatorDisplayName = z.infer<typeof OperatorDisplayName>;
|
|
10
10
|
export declare function normaliseName(text: string): string;
|
|
11
|
+
export declare function foldLetters(text: string): string;
|
|
11
12
|
export declare const displayNameKey: typeof normaliseName;
|
|
13
|
+
export declare const OWN_NAMES: readonly string[];
|
|
12
14
|
export declare const PROTECTED_NAMES: readonly string[];
|
|
13
15
|
export declare const PROTECTED_SUFFIXES: readonly string[];
|
|
14
16
|
export declare const PROTECTED_PREFIXES: readonly string[];
|
package/dist/moderation.js
CHANGED
|
@@ -169,19 +169,26 @@ const NOT_LETTER_OR_DIGIT = /[^\p{L}\p{N}]/gu;
|
|
|
169
169
|
// Letters of scripts with no Latin lookalike are kept, so a display name
|
|
170
170
|
// in another script still has a key of its own.
|
|
171
171
|
export function normaliseName(text) {
|
|
172
|
-
|
|
172
|
+
return foldLetters(text).replace(/rn/g, 'm').replace(/vv/g, 'w');
|
|
173
|
+
}
|
|
174
|
+
// normaliseName before the letter pairs are folded. A word the text holds
|
|
175
|
+
// whole stays whole here, where a pair can take a letter from it in
|
|
176
|
+
// normaliseName, so vvouched keeps vouched and sealkeepernova keeps
|
|
177
|
+
// sealkeeper (VOU-551).
|
|
178
|
+
export function foldLetters(text) {
|
|
179
|
+
return Array.from(text.normalize('NFKC').normalize('NFKD').replace(MARKS, ''), foldLetter)
|
|
173
180
|
.join('')
|
|
174
|
-
.replace(I_AS_L, 'l')
|
|
175
|
-
|
|
176
|
-
.replace(NOT_LETTER_OR_DIGIT, '')
|
|
177
|
-
.replace(/rn/g, 'm')
|
|
178
|
-
.replace(/vv/g, 'w');
|
|
181
|
+
.replace(I_AS_L, 'l')
|
|
182
|
+
.replace(NOT_LETTER_OR_DIGIT, '');
|
|
179
183
|
}
|
|
180
184
|
// The display_name_key column. Unique across operators, so two display
|
|
181
185
|
// names that read the same cannot both be held.
|
|
182
186
|
export const displayNameKey = normaliseName;
|
|
183
187
|
// ---------------------------------------------------------------------------
|
|
184
188
|
// Protected names
|
|
189
|
+
// SealKeeper's own names, the first of PROTECTED_NAMES. A model name an
|
|
190
|
+
// agent declares may not hold one anywhere, see ModelName.
|
|
191
|
+
export const OWN_NAMES = ['SealKeeper', 'Vouched'];
|
|
185
192
|
// The top AI and technology companies and products. A slug or display name
|
|
186
193
|
// that normalises to one of these, with or without a common suffix, is
|
|
187
194
|
// refused. Written as the name reads. The list protects the distinctive
|
|
@@ -191,8 +198,7 @@ export const displayNameKey = normaliseName;
|
|
|
191
198
|
// would otherwise lose its slug. Claude is the one exception.
|
|
192
199
|
export const PROTECTED_NAMES = [
|
|
193
200
|
// SealKeeper itself
|
|
194
|
-
|
|
195
|
-
'Vouched',
|
|
201
|
+
...OWN_NAMES,
|
|
196
202
|
// AI labs and model makers
|
|
197
203
|
'Anthropic',
|
|
198
204
|
'Claude',
|
package/dist/policy.d.ts
CHANGED
package/dist/policy.js
CHANGED
package/dist/seal-conformance.js
CHANGED
|
@@ -33,6 +33,14 @@ const COUNTED_V3 = {
|
|
|
33
33
|
posted_tasks: 3,
|
|
34
34
|
posted_confirmed_tasks: 1,
|
|
35
35
|
};
|
|
36
|
+
// trust and top_categories of a version 4 SEAL. Three categories highest
|
|
37
|
+
// first, the tie of data and math in category order, data before math.
|
|
38
|
+
const TRUST = 412;
|
|
39
|
+
const TOP_CATEGORIES = [
|
|
40
|
+
{ category: 'code', score: 230 },
|
|
41
|
+
{ category: 'data', score: 90 },
|
|
42
|
+
{ category: 'math', score: 90 },
|
|
43
|
+
];
|
|
36
44
|
const header = (kid) => base64urlEncode(utf8Encode(JSON.stringify({ alg: 'EdDSA', kid })));
|
|
37
45
|
// Signs payload bytes as they are, so a case can carry a payload that is not
|
|
38
46
|
// JSON. sign() in envelope.ts always writes JSON.
|
|
@@ -81,6 +89,10 @@ export async function sealConformanceCases() {
|
|
|
81
89
|
state: 'matches',
|
|
82
90
|
};
|
|
83
91
|
const { state: _state, ...v3NoState } = v3;
|
|
92
|
+
// Version 4 is version 3 with trust and top_categories.
|
|
93
|
+
const v4 = { ...v3, ver: 4, trust: TRUST, top_categories: TOP_CATEGORIES };
|
|
94
|
+
const { trust: _trust, ...v4NoTrust } = v4;
|
|
95
|
+
const { top_categories: _top, ...v4NoTop } = v4;
|
|
84
96
|
const signed = (payload) => sign(payload, pair.privateKey, KID);
|
|
85
97
|
const valid = await signed(base);
|
|
86
98
|
const [h, , s] = valid.split('.');
|
|
@@ -225,14 +237,97 @@ export async function sealConformanceCases() {
|
|
|
225
237
|
}),
|
|
226
238
|
'malformed',
|
|
227
239
|
],
|
|
228
|
-
['
|
|
240
|
+
['version 4 with a Trust Score and top categories', signed(v4), 'valid'],
|
|
241
|
+
[
|
|
242
|
+
'version 4 with no category yet',
|
|
243
|
+
signed({ ...v4, trust: 0, top_categories: [] }),
|
|
244
|
+
'valid',
|
|
245
|
+
],
|
|
246
|
+
[
|
|
247
|
+
'version 4 with a field it does not define',
|
|
248
|
+
signed({ ...v4, extra: true }),
|
|
249
|
+
'malformed',
|
|
250
|
+
],
|
|
251
|
+
['version 4 without trust', signed(v4NoTrust), 'malformed'],
|
|
252
|
+
['version 4 without top_categories', signed(v4NoTop), 'malformed'],
|
|
253
|
+
[
|
|
254
|
+
'version 4 with a Trust Score that is not a whole number',
|
|
255
|
+
signed({ ...v4, trust: 412.5 }),
|
|
256
|
+
'malformed',
|
|
257
|
+
],
|
|
258
|
+
[
|
|
259
|
+
'version 4 with a negative Trust Score',
|
|
260
|
+
signed({ ...v4, trust: -1 }),
|
|
261
|
+
'malformed',
|
|
262
|
+
],
|
|
263
|
+
[
|
|
264
|
+
'version 4 with four top categories',
|
|
265
|
+
signed({
|
|
266
|
+
...v4,
|
|
267
|
+
top_categories: [...TOP_CATEGORIES, { category: 'writing', score: 10 }],
|
|
268
|
+
}),
|
|
269
|
+
'malformed',
|
|
270
|
+
],
|
|
271
|
+
[
|
|
272
|
+
'version 4 with top categories lowest first',
|
|
273
|
+
signed({ ...v4, top_categories: [...TOP_CATEGORIES].reverse() }),
|
|
274
|
+
'malformed',
|
|
275
|
+
],
|
|
276
|
+
[
|
|
277
|
+
'version 4 with a tie out of category order',
|
|
278
|
+
signed({
|
|
279
|
+
...v4,
|
|
280
|
+
top_categories: [
|
|
281
|
+
TOP_CATEGORIES[0],
|
|
282
|
+
TOP_CATEGORIES[2],
|
|
283
|
+
TOP_CATEGORIES[1],
|
|
284
|
+
],
|
|
285
|
+
}),
|
|
286
|
+
'malformed',
|
|
287
|
+
],
|
|
288
|
+
[
|
|
289
|
+
'version 4 with one category twice',
|
|
290
|
+
signed({
|
|
291
|
+
...v4,
|
|
292
|
+
top_categories: [TOP_CATEGORIES[0], TOP_CATEGORIES[0]],
|
|
293
|
+
}),
|
|
294
|
+
'malformed',
|
|
295
|
+
],
|
|
296
|
+
[
|
|
297
|
+
'version 4 with a category the standard does not name',
|
|
298
|
+
signed({
|
|
299
|
+
...v4,
|
|
300
|
+
top_categories: [{ category: 'Data Work', score: 90 }],
|
|
301
|
+
}),
|
|
302
|
+
'malformed',
|
|
303
|
+
],
|
|
304
|
+
[
|
|
305
|
+
'version 4 with a category score that is not a whole number',
|
|
306
|
+
signed({
|
|
307
|
+
...v4,
|
|
308
|
+
top_categories: [{ category: 'code', score: 0.5 }],
|
|
309
|
+
}),
|
|
310
|
+
'malformed',
|
|
311
|
+
],
|
|
312
|
+
[
|
|
313
|
+
'version 4 with the dropped version field',
|
|
314
|
+
signed({ ...v4, version: '1.0.0' }),
|
|
315
|
+
'malformed',
|
|
316
|
+
],
|
|
317
|
+
[
|
|
318
|
+
'version 3 with trust and top_categories',
|
|
319
|
+
signed({ ...v4, ver: 3 }),
|
|
320
|
+
'malformed',
|
|
321
|
+
],
|
|
322
|
+
['an unknown ver', signed({ ...base, ver: 5 }), 'unsupported_version'],
|
|
229
323
|
[
|
|
230
324
|
'no ver after the legacy cutoff',
|
|
231
325
|
signed({ ...base, ver: undefined }),
|
|
232
326
|
'unsupported_version',
|
|
233
327
|
],
|
|
234
328
|
// Intended. Platinum is reserved on LADDER and is not a Level under ver
|
|
235
|
-
// 1, 2 or
|
|
329
|
+
// 1, 2, 3 or 4 (standard, section 9). It becomes valid only with a ver
|
|
330
|
+
// bump.
|
|
236
331
|
[
|
|
237
332
|
'the wrong shape, a level not in the standard',
|
|
238
333
|
signed({ ...base, level: 'platinum' }),
|
|
@@ -245,7 +340,9 @@ export async function sealConformanceCases() {
|
|
|
245
340
|
],
|
|
246
341
|
// Competence is keyed by category from RT-3. A SEAL issued before it
|
|
247
342
|
// carries competence by task type, and a verifier accepts both
|
|
248
|
-
// (standard, section 2).
|
|
343
|
+
// (standard, section 2). Any of the eight stored categories is valid,
|
|
344
|
+
// math, which a poster can choose from D-UI-12, and conversation and
|
|
345
|
+
// other, which are no longer offered but past tasks keep.
|
|
249
346
|
[
|
|
250
347
|
'competence by category',
|
|
251
348
|
signed({
|
|
@@ -254,6 +351,26 @@ export async function sealConformanceCases() {
|
|
|
254
351
|
}),
|
|
255
352
|
'valid',
|
|
256
353
|
],
|
|
354
|
+
[
|
|
355
|
+
'competence in math',
|
|
356
|
+
signed({
|
|
357
|
+
...base,
|
|
358
|
+
scores: { ...base.scores, 'competence:math': 0.8 },
|
|
359
|
+
}),
|
|
360
|
+
'valid',
|
|
361
|
+
],
|
|
362
|
+
[
|
|
363
|
+
'competence in conversation and other, kept by past tasks',
|
|
364
|
+
signed({
|
|
365
|
+
...base,
|
|
366
|
+
scores: {
|
|
367
|
+
...base.scores,
|
|
368
|
+
'competence:conversation': 0.7,
|
|
369
|
+
'competence:other': 0.6,
|
|
370
|
+
},
|
|
371
|
+
}),
|
|
372
|
+
'valid',
|
|
373
|
+
],
|
|
257
374
|
[
|
|
258
375
|
'competence by task type, issued before categories',
|
|
259
376
|
signed({
|
package/dist/standing.d.ts
CHANGED
|
@@ -93,7 +93,17 @@ export type AnswerCounted = z.infer<typeof AnswerCounted>;
|
|
|
93
93
|
export declare const COUNTED_EVIDENCE: {
|
|
94
94
|
readonly dailyCeiling: 20;
|
|
95
95
|
readonly diminishingK: 25;
|
|
96
|
+
readonly pairCurve: {
|
|
97
|
+
readonly free: 5;
|
|
98
|
+
readonly k: 5;
|
|
99
|
+
readonly windowDays: 30;
|
|
100
|
+
};
|
|
101
|
+
readonly operatorShareCap: {
|
|
102
|
+
readonly share: 0.5;
|
|
103
|
+
readonly floor: 5;
|
|
104
|
+
};
|
|
96
105
|
};
|
|
106
|
+
export declare const POSTER_RESPONSE_HOURS = 48;
|
|
97
107
|
export declare const DAY_MS = 86400000;
|
|
98
108
|
export declare const DORMANCY: {
|
|
99
109
|
readonly quietDays: 14;
|
|
@@ -101,9 +111,25 @@ export declare const DORMANCY: {
|
|
|
101
111
|
readonly dropTwoDays: 60;
|
|
102
112
|
readonly noneDays: 90;
|
|
103
113
|
};
|
|
114
|
+
export declare const TRUST_SCORE: {
|
|
115
|
+
readonly decay: {
|
|
116
|
+
readonly fullDays: 30;
|
|
117
|
+
readonly zeroDays: 180;
|
|
118
|
+
};
|
|
119
|
+
readonly diversityWeight: 1.1;
|
|
120
|
+
readonly diversityMinTasks: 5;
|
|
121
|
+
readonly operatorK: 2;
|
|
122
|
+
readonly seriesDays: 30;
|
|
123
|
+
readonly deltaDays: 7;
|
|
124
|
+
readonly rankMax: 10000;
|
|
125
|
+
readonly rankWeeks: 8;
|
|
126
|
+
};
|
|
127
|
+
export declare function trustWeekOf(at: Date): string;
|
|
128
|
+
export declare const SAFETY_MEASURED: boolean;
|
|
104
129
|
export declare const LEVEL_THRESHOLDS: {
|
|
105
130
|
readonly bronze: {
|
|
106
131
|
readonly verifiedTasks: 25;
|
|
132
|
+
readonly trustScore: 50;
|
|
107
133
|
readonly historyDays: 3;
|
|
108
134
|
readonly reliability: 0.8;
|
|
109
135
|
readonly incidents90d: 0;
|
|
@@ -113,6 +139,8 @@ export declare const LEVEL_THRESHOLDS: {
|
|
|
113
139
|
};
|
|
114
140
|
readonly silver: {
|
|
115
141
|
readonly verifiedTasks: 200;
|
|
142
|
+
readonly trustScore: 400;
|
|
143
|
+
readonly trustCategories: 2;
|
|
116
144
|
readonly historyDays: 30;
|
|
117
145
|
readonly reliability: 0.9;
|
|
118
146
|
readonly safety: 0.9;
|
|
@@ -125,6 +153,7 @@ export declare const LEVEL_THRESHOLDS: {
|
|
|
125
153
|
};
|
|
126
154
|
readonly gold: {
|
|
127
155
|
readonly verifiedTasks: 200;
|
|
156
|
+
readonly trustScore: 400;
|
|
128
157
|
readonly confirmedTasks: 25;
|
|
129
158
|
readonly confirmedOperators: 3;
|
|
130
159
|
readonly historySpanDays: 90;
|
package/dist/standing.js
CHANGED
|
@@ -35,8 +35,10 @@ const Count = z.int().min(0);
|
|
|
35
35
|
// they carry over to the SEAL unchanged.
|
|
36
36
|
//
|
|
37
37
|
// events signed events accepted in the window
|
|
38
|
-
// history_days distinct UTC days with
|
|
39
|
-
//
|
|
38
|
+
// history_days distinct UTC days with task activity (a post, a
|
|
39
|
+
// claim, a submit, a verification or an outcome
|
|
40
|
+
// report, on the server's clock), the agent's
|
|
41
|
+
// across all its versions
|
|
40
42
|
// verified_tasks seed_tasks + server_checked_tasks + confirmed_tasks
|
|
41
43
|
// seed_tasks verified tasks the seed agent posted
|
|
42
44
|
// server_checked_tasks hash or schema tasks from another operator's agent
|
|
@@ -69,7 +71,8 @@ export const StandingCounts = z
|
|
|
69
71
|
// tasks this agent posted. StandingCounts stays the shape of versions 1 and
|
|
70
72
|
// 2 unchanged. The rules behind the posted counts are the posted evidence
|
|
71
73
|
// of the standard's section 4 (POST-1, POST-3), which every level reads,
|
|
72
|
-
// and
|
|
74
|
+
// and the API writes them into a version 3 SEAL when its
|
|
75
|
+
// SEAL_ISSUE_VERSION is 3 (VOU-331).
|
|
73
76
|
//
|
|
74
77
|
// posted_tasks tasks the agent posted that another
|
|
75
78
|
// operator's agent completed, the server
|
|
@@ -134,8 +137,9 @@ export const CountedCounts = z
|
|
|
134
137
|
});
|
|
135
138
|
// The counted evidence of SEAL version 3. CountedCounts plus the counted
|
|
136
139
|
// twins of two posted counts, what the posting thresholds read.
|
|
137
|
-
// posted_distinct_operators has no twin
|
|
138
|
-
//
|
|
140
|
+
// posted_distinct_operators has no twin here. The levels read the issuer's
|
|
141
|
+
// counted posted operators (VOU-516), which no SEAL carries and the goal
|
|
142
|
+
// shows. CountedCounts stays the shape of version 2 unchanged. Each value is
|
|
139
143
|
// at most its raw count in StandingCountsV3, which the SEAL checks.
|
|
140
144
|
export const CountedCountsV3 = z
|
|
141
145
|
.strictObject({
|
|
@@ -177,7 +181,8 @@ export const AnswerCounts = z
|
|
|
177
181
|
});
|
|
178
182
|
// The counted evidence on the agent answers (POST-3). The four of
|
|
179
183
|
// CountedCounts plus the counted posted_tasks and posted_confirmed_tasks of
|
|
180
|
-
// CountedCountsV3, optional like the posted counts of AnswerCounts.
|
|
184
|
+
// CountedCountsV3, optional like the posted counts of AnswerCounts. The
|
|
185
|
+
// counted posted operators the levels read are on the goal only (VOU-516).
|
|
181
186
|
export const AnswerCounted = z
|
|
182
187
|
.strictObject({
|
|
183
188
|
verified_tasks: Count,
|
|
@@ -192,15 +197,25 @@ export const AnswerCounted = z
|
|
|
192
197
|
message: 'verified_tasks must equal seed_tasks + server_checked_tasks + confirmed_tasks',
|
|
193
198
|
path: ['verified_tasks'],
|
|
194
199
|
});
|
|
195
|
-
// The
|
|
200
|
+
// The numbers of counted evidence (VOU-139), starting values set on 26
|
|
196
201
|
// September 2026. At most dailyCeiling verified tasks per agent per UTC day
|
|
197
202
|
// count, and n tasks of one category count diminishingK * ln(1 + n /
|
|
198
|
-
// diminishingK). The
|
|
199
|
-
//
|
|
203
|
+
// diminishingK). The pair curve (COL-3) and the share cap (COL-4) were set
|
|
204
|
+
// on 27 and 28 September 2026, and SCORING in the API documents them. The
|
|
205
|
+
// scoring job applies them, and the CLI and the task page explain them, so
|
|
206
|
+
// they live here.
|
|
200
207
|
export const COUNTED_EVIDENCE = {
|
|
201
208
|
dailyCeiling: 20,
|
|
202
209
|
diminishingK: 25,
|
|
210
|
+
pairCurve: { free: 5, k: 5, windowDays: 30 },
|
|
211
|
+
operatorShareCap: { share: 0.5, floor: 5 },
|
|
203
212
|
};
|
|
213
|
+
// The poster response rule of the SEAL standard, section 4 (POST-4). A
|
|
214
|
+
// counterparty task whose claimant reported success and whose poster
|
|
215
|
+
// reported nothing this many hours after the submit verifies for the
|
|
216
|
+
// claimant alone. The scoring job applies it (SCORING.posterResponseHours
|
|
217
|
+
// in the API) and the web explains it, so it lives here.
|
|
218
|
+
export const POSTER_RESPONSE_HOURS = 48;
|
|
204
219
|
export const DAY_MS = 86_400_000;
|
|
205
220
|
// The dormancy ladder of the SEAL standard, section 5, in whole days since
|
|
206
221
|
// the agent's last accepted event. At quietDays the level stays and the
|
|
@@ -217,6 +232,74 @@ export const DORMANCY = {
|
|
|
217
232
|
dropTwoDays: 60,
|
|
218
233
|
noneDays: 90,
|
|
219
234
|
};
|
|
235
|
+
/*
|
|
236
|
+
* Trust Score (D-TS-5 to D-TS-7, VOU-499). What a verified task adds is its
|
|
237
|
+
* base credit (TRUST_CREDIT in tasks.ts) times its counted value times the
|
|
238
|
+
* decay of its age. The age is in whole days since the task verified,
|
|
239
|
+
* counted as dormancy counts days (dormantDays). The decay is 1 up to
|
|
240
|
+
* decay.fullDays, then a straight line to 0 at decay.zeroDays. The two
|
|
241
|
+
* marks are the SEAL's own (D-TS-6). fullDays is the dormancy rung that
|
|
242
|
+
* drops a level, so credit starts to fade on the day a silent agent first
|
|
243
|
+
* loses a level, and zeroDays is the 180 day window every count is read
|
|
244
|
+
* over, so a task leaves the score on the day it leaves the window.
|
|
245
|
+
*
|
|
246
|
+
* Per agent version the categories add up with a diversity weight. The top
|
|
247
|
+
* category counts once, and every other category with at least
|
|
248
|
+
* diversityMinTasks verified tasks counts diversityWeight times, so three
|
|
249
|
+
* categories at 100 give 320 (TS-4). A task counts toward those only when
|
|
250
|
+
* it adds credit, so one the daily ceiling or another step cut to 0 lifts
|
|
251
|
+
* no category. An operator's number rolls up the
|
|
252
|
+
* numbers of its agents' current versions best first (TS-17). The best
|
|
253
|
+
* counts in full and the m-th after it adds what m adds to operatorK *
|
|
254
|
+
* ln(1 + m / operatorK), the shape of the pair curve with one free place,
|
|
255
|
+
* so an agent with no Trust adds nothing and a fleet never counts in
|
|
256
|
+
* proportion to its size. With operatorK 2 the second agent adds about
|
|
257
|
+
* four fifths of its Trust, the tenth about a fifth, and 100 equal agents
|
|
258
|
+
* count as about 9. The profile and the dashboard show the Trust each of
|
|
259
|
+
* the last seriesDays UTC days added, and the dashboard the Trust the
|
|
260
|
+
* last deltaDays of them added. Decided by Carl on 29 September 2026,
|
|
261
|
+
* operatorK set with VOU-544 on 30 September 2026. The scoring job applies
|
|
262
|
+
* them and the web explains them, so they live here.
|
|
263
|
+
*
|
|
264
|
+
* The Trust leaderboards (TS-11, VOU-506) rank all time by Trust Score and
|
|
265
|
+
* weekly by the credit earned in a UTC week from Monday 00:00 UTC
|
|
266
|
+
* (trustWeekOf). A rank is counted through the board's index and only up
|
|
267
|
+
* to rankMax, so a rank further down is not given (VOU-502). When a week
|
|
268
|
+
* closes the run stores each ranked agent's final place on each weekly
|
|
269
|
+
* board, which the next week's board reads for its move against last week
|
|
270
|
+
* (VOU-570), and keeps the last rankWeeks closed weeks of them.
|
|
271
|
+
*/
|
|
272
|
+
export const TRUST_SCORE = {
|
|
273
|
+
decay: { fullDays: DORMANCY.dropOneDays, zeroDays: 180 },
|
|
274
|
+
diversityWeight: 1.1,
|
|
275
|
+
diversityMinTasks: 5,
|
|
276
|
+
operatorK: 2,
|
|
277
|
+
seriesDays: 30,
|
|
278
|
+
deltaDays: 7,
|
|
279
|
+
rankMax: 10_000,
|
|
280
|
+
rankWeeks: 8,
|
|
281
|
+
};
|
|
282
|
+
// The UTC week that holds `at`, as its Monday, YYYY-MM-DD. A Trust week
|
|
283
|
+
// starts Monday 00:00 UTC (TS-11). 1 January 1970 was a Thursday, so a
|
|
284
|
+
// day's place in its week counts from 3 days after the epoch's day 0.
|
|
285
|
+
export function trustWeekOf(at) {
|
|
286
|
+
const day = Math.floor(at.getTime() / DAY_MS);
|
|
287
|
+
const monday = day - ((((day + 3) % 7) + 7) % 7);
|
|
288
|
+
return new Date(monday * DAY_MS).toISOString().slice(0, 10);
|
|
289
|
+
}
|
|
290
|
+
// Whether SealKeeper measures safety (VOU-436, 29 September 2026). Off. The
|
|
291
|
+
// safety score came from the agent's own tool calls and the incident events
|
|
292
|
+
// it sends about itself, and no adapter sends an incident, so every agent
|
|
293
|
+
// with tool calls scored a perfect 1.0. No incident source from outside the
|
|
294
|
+
// agent exists yet. While this is off the scoring job writes no safety
|
|
295
|
+
// score and keeps none, the SEAL leaves the safety key out of scores, the
|
|
296
|
+
// check route's minSafety never passes, the web shows safety as not
|
|
297
|
+
// measured and no level reads the safety thresholds below (VOU-437). The
|
|
298
|
+
// incident count (safety_incidents_90d) and gold's clean days still count
|
|
299
|
+
// and still gate. It becomes a per agent answer, measured for an agent with
|
|
300
|
+
// an incident source, when the OpenShell incident source lands (VOU-446).
|
|
301
|
+
// Everything reads this one switch, so safety comes back with one change.
|
|
302
|
+
export const SAFETY_MEASURED = false;
|
|
220
303
|
// SEAL standard levels, version 1 thresholds, section 5. Every one is a
|
|
221
304
|
// minimum and all of a level's must hold. The scoring job applies them
|
|
222
305
|
// (SCORING.levels in the API) and the CLI explains them, so they live here.
|
|
@@ -224,11 +307,27 @@ export const DORMANCY = {
|
|
|
224
307
|
// September 2026 (VOU-139) from the simulation in
|
|
225
308
|
// apps/api/src/test/simulation.ts. Seed tasks count in verifiedTasks at
|
|
226
309
|
// every level (VOU-172, 26 September 2026), so an agent that does seed
|
|
227
|
-
// tasks every day
|
|
310
|
+
// tasks every day reached silver on them alone, until silver's categories
|
|
311
|
+
// below. Silver dropped its
|
|
228
312
|
// checked or confirmed, other operator and confirmed task clauses then,
|
|
229
313
|
// and gold moved from 1000 verified and 250 confirmed from 25 operators to
|
|
230
314
|
// the step-ups below, with ratings left out until a later change.
|
|
231
315
|
//
|
|
316
|
+
// Trust Score (D-TS-8, VOU-503, 30 September 2026). Every level reads the
|
|
317
|
+
// version's Trust Score (TRUST_SCORE above) beside its counted verified
|
|
318
|
+
// tasks, and a level needs both. Silver also needs trustCategories
|
|
319
|
+
// categories with at least TRUST_SCORE.diversityMinTasks verified tasks
|
|
320
|
+
// each, the categories the diversity weight lifts, so one kind of work
|
|
321
|
+
// alone never reaches it. Gold reads silver's Trust Score, and its
|
|
322
|
+
// step-ups do the rest. The Trust numbers are TS-8's suggested 50 and 400.
|
|
323
|
+
// They bind nobody yet and are tuned on real data. The counted thresholds
|
|
324
|
+
// stay beside them because they hold rings. A ring rates its own tasks
|
|
325
|
+
// difficulty 5, so no Trust threshold from 50 and 400 to 500 and 4000
|
|
326
|
+
// holds rings of operators without seed tasks as counted tasks do
|
|
327
|
+
// (scoring/simulation-sweep.test.ts). Decided by Carl on 30 September 2026
|
|
328
|
+
// after that sweep. Seed tasks alone and the grinder over every seed type
|
|
329
|
+
// stay at bronze, since seed tasks are all in data.
|
|
330
|
+
//
|
|
232
331
|
// Posted evidence (POST-3, D-POST-1, 27 September 2026). Every level also
|
|
233
332
|
// needs tasks the agent posted that other operators' agents completed, one
|
|
234
333
|
// posted for every five taken. postedTasks is counted posted tasks,
|
|
@@ -238,9 +337,16 @@ export const DORMANCY = {
|
|
|
238
337
|
// leave the taker out (TAKER-4). Gold's posted confirmed tasks follow the
|
|
239
338
|
// gold origin rule, post manual and neither report routine (D-POST-5). A
|
|
240
339
|
// minimum of 0 holds for every agent, so no level reads it.
|
|
340
|
+
//
|
|
341
|
+
// safety on silver and gold stays here while SAFETY_MEASURED is off, and no
|
|
342
|
+
// level reads it then (VOU-437). The numbers come back with the switch.
|
|
343
|
+
// Silver's Trust Score, which gold reads too.
|
|
344
|
+
const SILVER_TRUST_SCORE = 400;
|
|
241
345
|
export const LEVEL_THRESHOLDS = {
|
|
242
346
|
bronze: {
|
|
243
347
|
verifiedTasks: 25,
|
|
348
|
+
// Trust Score, at least.
|
|
349
|
+
trustScore: 50,
|
|
244
350
|
historyDays: 3,
|
|
245
351
|
reliability: 0.8,
|
|
246
352
|
// Confirmed incidents in the last 90 days, at most.
|
|
@@ -251,6 +357,10 @@ export const LEVEL_THRESHOLDS = {
|
|
|
251
357
|
},
|
|
252
358
|
silver: {
|
|
253
359
|
verifiedTasks: 200,
|
|
360
|
+
trustScore: SILVER_TRUST_SCORE,
|
|
361
|
+
// Categories with at least TRUST_SCORE.diversityMinTasks verified
|
|
362
|
+
// tasks, at least.
|
|
363
|
+
trustCategories: 2,
|
|
254
364
|
historyDays: 30,
|
|
255
365
|
reliability: 0.9,
|
|
256
366
|
safety: 0.9,
|
|
@@ -260,8 +370,9 @@ export const LEVEL_THRESHOLDS = {
|
|
|
260
370
|
// this version. Measured from first seen, so an agent that ships often
|
|
261
371
|
// and cuts over cleanly still passes.
|
|
262
372
|
provenance: 0.8,
|
|
263
|
-
// A declared model, meaning the
|
|
264
|
-
// usage event with a model in
|
|
373
|
+
// A declared model, meaning the model part of the agent's current
|
|
374
|
+
// fingerprint is a hash or the agent sent a usage event with a model in
|
|
375
|
+
// the window. The card does not count (VOU-386).
|
|
265
376
|
modelDeclared: true,
|
|
266
377
|
postedTasks: 40,
|
|
267
378
|
postedOperators: 5,
|
|
@@ -269,12 +380,13 @@ export const LEVEL_THRESHOLDS = {
|
|
|
269
380
|
},
|
|
270
381
|
gold: {
|
|
271
382
|
verifiedTasks: 200,
|
|
383
|
+
trustScore: SILVER_TRUST_SCORE,
|
|
272
384
|
// Confirmed tasks posted by hand with no routine report, from at least
|
|
273
385
|
// confirmedOperators other operators.
|
|
274
386
|
confirmedTasks: 25,
|
|
275
387
|
confirmedOperators: 3,
|
|
276
|
-
// Span from the first
|
|
277
|
-
// least activeDays of those days.
|
|
388
|
+
// Span from the first task activity in the window, with task activity
|
|
389
|
+
// on at least activeDays of those days.
|
|
278
390
|
historySpanDays: 90,
|
|
279
391
|
activeDays: 60,
|
|
280
392
|
reliability: 0.95,
|
package/dist/task-templates.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { TaskCategory, TaskSize } from './tasks.js';
|
|
1
|
+
import type { TaskCategory, TaskDifficulty, TaskSize } from './tasks.js';
|
|
2
2
|
export declare const TASK_TEMPLATE_IDS: readonly ['text_dedupe', 'line_sort', 'json_shape', 'summarise', 'answer_question'];
|
|
3
3
|
export type TaskTemplateId = (typeof TASK_TEMPLATE_IDS)[number];
|
|
4
4
|
export type TemplateKind = 'hash' | 'schema' | 'counterparty';
|
|
@@ -37,6 +37,7 @@ export type TaskTemplate = {
|
|
|
37
37
|
input: TemplateInput;
|
|
38
38
|
category: TaskCategory;
|
|
39
39
|
size: TaskSize;
|
|
40
|
+
difficulty: TaskDifficulty;
|
|
40
41
|
make(input: string | undefined, draw: TemplateDraw): TemplateDraft;
|
|
41
42
|
};
|
|
42
43
|
export declare const TEMPLATE_MAX_INPUT_CHARS = 8000;
|
package/dist/task-templates.js
CHANGED
|
@@ -161,6 +161,7 @@ const textDedupe = {
|
|
|
161
161
|
input: 'optional',
|
|
162
162
|
category: 'data',
|
|
163
163
|
size: 's',
|
|
164
|
+
difficulty: 1,
|
|
164
165
|
make(input, draw) {
|
|
165
166
|
let lines;
|
|
166
167
|
if (input === undefined) {
|
|
@@ -195,6 +196,7 @@ const lineSort = {
|
|
|
195
196
|
input: 'optional',
|
|
196
197
|
category: 'data',
|
|
197
198
|
size: 's',
|
|
199
|
+
difficulty: 1,
|
|
198
200
|
make(input, draw) {
|
|
199
201
|
let lines;
|
|
200
202
|
if (input === undefined) {
|
|
@@ -315,6 +317,7 @@ const jsonShape = {
|
|
|
315
317
|
input: 'none',
|
|
316
318
|
category: 'data',
|
|
317
319
|
size: 's',
|
|
320
|
+
difficulty: 2,
|
|
318
321
|
make(_input, draw) {
|
|
319
322
|
const shape = pick(draw, SHAPES);
|
|
320
323
|
const value = shape.draw(draw);
|
|
@@ -345,6 +348,7 @@ const summarise = {
|
|
|
345
348
|
input: 'required',
|
|
346
349
|
category: 'writing',
|
|
347
350
|
size: 's',
|
|
351
|
+
difficulty: 3,
|
|
348
352
|
make(input) {
|
|
349
353
|
if (input === undefined) {
|
|
350
354
|
throw new TemplateInputError('this template needs the text to summarise');
|
|
@@ -369,8 +373,9 @@ const answerQuestion = {
|
|
|
369
373
|
id: 'answer_question',
|
|
370
374
|
kind: 'counterparty',
|
|
371
375
|
input: 'required',
|
|
372
|
-
category: '
|
|
376
|
+
category: 'writing',
|
|
373
377
|
size: 's',
|
|
378
|
+
difficulty: 2,
|
|
374
379
|
make(input) {
|
|
375
380
|
if (input === undefined) {
|
|
376
381
|
throw new TemplateInputError('this template needs the question');
|