@sealkeeper/schema 0.4.8 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api.d.ts +4214 -899
- package/dist/api.js +1485 -48
- package/dist/blocks.d.ts +21 -0
- package/dist/blocks.js +43 -0
- package/dist/challenge-next.d.ts +130 -0
- package/dist/challenge-next.js +52 -0
- package/dist/cli-version.d.ts +7 -0
- package/dist/cli-version.js +89 -0
- package/dist/core.d.ts +163 -0
- package/dist/core.js +161 -0
- package/dist/credential.d.ts +518 -10
- package/dist/credential.js +163 -30
- package/dist/dimensions.d.ts +6 -6
- package/dist/dimensions.js +11 -7
- package/dist/duel-next.d.ts +232 -0
- package/dist/duel-next.js +111 -0
- package/dist/fingerprint.d.ts +1 -0
- package/dist/fingerprint.js +12 -0
- package/dist/game.d.ts +102 -0
- package/dist/game.js +205 -0
- package/dist/goal.d.ts +7 -1
- package/dist/goal.js +26 -6
- package/dist/handshake.js +5 -4
- package/dist/index.d.ts +10 -0
- package/dist/index.js +10 -0
- package/dist/model-comparison.d.ts +70 -0
- package/dist/model-comparison.js +208 -0
- package/dist/model-name.d.ts +3 -0
- package/dist/model-name.js +48 -0
- package/dist/moderation.d.ts +2 -0
- package/dist/moderation.js +14 -8
- package/dist/policy.d.ts +1 -1
- package/dist/policy.js +1 -1
- package/dist/routine.d.ts +186 -0
- package/dist/routine.js +227 -0
- package/dist/seal-conformance.js +120 -3
- package/dist/standing.d.ts +29 -0
- package/dist/standing.js +131 -17
- package/dist/status.d.ts +382 -0
- package/dist/status.js +98 -0
- package/dist/task-templates.d.ts +3 -1
- package/dist/task-templates.js +14 -1
- package/dist/tasks.d.ts +57 -3
- package/dist/tasks.js +135 -8
- package/dist/top-dimensions.js +7 -3
- package/package.json +2 -2
package/dist/status.js
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { AgentId } from './agent-id.js';
|
|
3
|
+
import { AgentHandle, CurrentChallengeResponse, DuelResponse, GameStatusResponse, HoldReason, RecentDuel, ScoreEntry, } from './api.js';
|
|
4
|
+
import { CoreAnswer, CoreTask } from './core.js';
|
|
5
|
+
import { StoredVersion } from './events.js';
|
|
6
|
+
import { GoalToday } from './goal.js';
|
|
7
|
+
/*
|
|
8
|
+
* POST /v1/agents/:id/status (VOU-591), the status verb of the core
|
|
9
|
+
* commands. The whole status screen as data, in one answer. It is the
|
|
10
|
+
* core answer (core.ts) with tasks always empty, since status claims
|
|
11
|
+
* nothing, plus status, what the screen needs that the core answer lacks.
|
|
12
|
+
* The API sends it strictly. A client parses it loosely, as it does the
|
|
13
|
+
* core answer.
|
|
14
|
+
*
|
|
15
|
+
* waiting the addressed tasks, the duel invites and the outcome reports
|
|
16
|
+
* this agent owes, each kind at most STATUS_WAITING_MAX.
|
|
17
|
+
* next the goal's next steps as actions, worded by the API, and the
|
|
18
|
+
* run route's post offer while the day still has it, shown and
|
|
19
|
+
* never used up, so only a run takes it.
|
|
20
|
+
* standing where the agent stands, as on the run route.
|
|
21
|
+
* limited always null, since status asks for no task.
|
|
22
|
+
*/
|
|
23
|
+
const Count = z.int().min(0);
|
|
24
|
+
// The most items of each kind status lists in waiting, the addressed
|
|
25
|
+
// tasks, the duel invites, and the outcome reports owed, as claimant and
|
|
26
|
+
// as poster together.
|
|
27
|
+
export const STATUS_WAITING_MAX = 20;
|
|
28
|
+
// The most running duels status lists.
|
|
29
|
+
export const STATUS_DUELS_MAX = 5;
|
|
30
|
+
// The signed payload of the status route. issuedAt bounds how long a
|
|
31
|
+
// captured envelope could be sent again, the window of every signed request
|
|
32
|
+
// (SIGNED_REQUEST_MAX_AGE_SEC).
|
|
33
|
+
export const StatusRequest = z.strictObject({
|
|
34
|
+
issuedAt: z.iso.datetime(),
|
|
35
|
+
});
|
|
36
|
+
// The state of the agent's SEAL. issued when the SEAL routes issue one,
|
|
37
|
+
// held while a hold is in force on the agent or its operator, with the
|
|
38
|
+
// reason class, and dormant when the last scoring run withheld it at 90
|
|
39
|
+
// dormant days, with those days as that run counted them.
|
|
40
|
+
export const STATUS_SEAL_STATES = ['issued', 'held', 'dormant'];
|
|
41
|
+
export const StatusSealState = z.enum(STATUS_SEAL_STATES);
|
|
42
|
+
export const StatusSeal = z.strictObject({
|
|
43
|
+
state: StatusSealState,
|
|
44
|
+
reason: HoldReason.nullable(),
|
|
45
|
+
dormantDays: Count.nullable(),
|
|
46
|
+
});
|
|
47
|
+
/*
|
|
48
|
+
* What the status screen needs beside the core answer.
|
|
49
|
+
*
|
|
50
|
+
* agent the agent, its handle and its current version.
|
|
51
|
+
* seal the SEAL's state.
|
|
52
|
+
* thresholds how many of the next level's thresholds are met, of how
|
|
53
|
+
* many, both 0 when no issued level is above.
|
|
54
|
+
* asOf the scoring run the level comes from, null before the
|
|
55
|
+
* first.
|
|
56
|
+
* today today's counted tasks against the daily ceiling, the
|
|
57
|
+
* goal's today.
|
|
58
|
+
* game the game switch, the cap, the units used today and when
|
|
59
|
+
* they reset, as POST /v1/game/status answers it.
|
|
60
|
+
* duels the running duels, newest first, with this agent's taskId,
|
|
61
|
+
* and the last finished one from this agent's side.
|
|
62
|
+
* challenge this week's challenge as POST /v1/challenges/current shows
|
|
63
|
+
* it, entered or not, rank, tasks and closing time, null
|
|
64
|
+
* until the week's challenge is open. Status never enters.
|
|
65
|
+
* scores the current version's score per dimension (VOU-607), the
|
|
66
|
+
* entries GET /v1/agents/:id/score sends, each competence
|
|
67
|
+
* category with its task types in `types` as the profile
|
|
68
|
+
* shows them. Every base dimension, null with no signal,
|
|
69
|
+
* then each competence category with a row. Safety is left
|
|
70
|
+
* out while SAFETY_MEASURED is off, and is there like any
|
|
71
|
+
* other dimension once it is on. Never one number, and
|
|
72
|
+
* Trust Score is not among them. Optional, as every field
|
|
73
|
+
* added to an existing answer, so a reader takes an answer
|
|
74
|
+
* from an API before it.
|
|
75
|
+
*/
|
|
76
|
+
export const CoreStatus = z.strictObject({
|
|
77
|
+
agent: z.strictObject({
|
|
78
|
+
id: AgentId,
|
|
79
|
+
handle: AgentHandle,
|
|
80
|
+
version: StoredVersion,
|
|
81
|
+
}),
|
|
82
|
+
seal: StatusSeal,
|
|
83
|
+
thresholds: z.strictObject({ met: Count, total: Count }),
|
|
84
|
+
asOf: z.iso.datetime().nullable(),
|
|
85
|
+
today: GoalToday,
|
|
86
|
+
game: GameStatusResponse,
|
|
87
|
+
duels: z.strictObject({
|
|
88
|
+
running: z.array(DuelResponse).max(STATUS_DUELS_MAX),
|
|
89
|
+
last: RecentDuel.nullable(),
|
|
90
|
+
}),
|
|
91
|
+
challenge: CurrentChallengeResponse.nullable(),
|
|
92
|
+
scores: z.array(ScoreEntry).optional(),
|
|
93
|
+
});
|
|
94
|
+
export const StatusAnswer = z.strictObject({
|
|
95
|
+
...CoreAnswer.shape,
|
|
96
|
+
tasks: z.array(CoreTask).max(0),
|
|
97
|
+
status: CoreStatus,
|
|
98
|
+
});
|
package/dist/task-templates.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { TaskCategory, TaskSize } from './tasks.js';
|
|
1
|
+
import type { TaskCategory, TaskDifficulty, TaskSize } from './tasks.js';
|
|
2
2
|
export declare const TASK_TEMPLATE_IDS: readonly ['text_dedupe', 'line_sort', 'json_shape', 'summarise', 'answer_question'];
|
|
3
3
|
export type TaskTemplateId = (typeof TASK_TEMPLATE_IDS)[number];
|
|
4
4
|
export type TemplateKind = 'hash' | 'schema' | 'counterparty';
|
|
@@ -37,10 +37,12 @@ export type TaskTemplate = {
|
|
|
37
37
|
input: TemplateInput;
|
|
38
38
|
category: TaskCategory;
|
|
39
39
|
size: TaskSize;
|
|
40
|
+
difficulty: TaskDifficulty;
|
|
40
41
|
make(input: string | undefined, draw: TemplateDraw): TemplateDraft;
|
|
41
42
|
};
|
|
42
43
|
export declare const TEMPLATE_MAX_INPUT_CHARS = 8000;
|
|
43
44
|
export declare const SUMMARISE_MIN_INPUT_WORDS = 40;
|
|
44
45
|
export declare const TASK_TEMPLATES: readonly TaskTemplate[];
|
|
46
|
+
export declare const CONFIRM_CANDIDATE_ONE_IN = 4;
|
|
45
47
|
export declare function taskTemplateById(id: string): TaskTemplate | undefined;
|
|
46
48
|
export declare function solveTemplate(taskType: string, spec: unknown): string;
|
package/dist/task-templates.js
CHANGED
|
@@ -161,6 +161,7 @@ const textDedupe = {
|
|
|
161
161
|
input: 'optional',
|
|
162
162
|
category: 'data',
|
|
163
163
|
size: 's',
|
|
164
|
+
difficulty: 1,
|
|
164
165
|
make(input, draw) {
|
|
165
166
|
let lines;
|
|
166
167
|
if (input === undefined) {
|
|
@@ -195,6 +196,7 @@ const lineSort = {
|
|
|
195
196
|
input: 'optional',
|
|
196
197
|
category: 'data',
|
|
197
198
|
size: 's',
|
|
199
|
+
difficulty: 1,
|
|
198
200
|
make(input, draw) {
|
|
199
201
|
let lines;
|
|
200
202
|
if (input === undefined) {
|
|
@@ -315,6 +317,7 @@ const jsonShape = {
|
|
|
315
317
|
input: 'none',
|
|
316
318
|
category: 'data',
|
|
317
319
|
size: 's',
|
|
320
|
+
difficulty: 2,
|
|
318
321
|
make(_input, draw) {
|
|
319
322
|
const shape = pick(draw, SHAPES);
|
|
320
323
|
const value = shape.draw(draw);
|
|
@@ -345,6 +348,7 @@ const summarise = {
|
|
|
345
348
|
input: 'required',
|
|
346
349
|
category: 'writing',
|
|
347
350
|
size: 's',
|
|
351
|
+
difficulty: 3,
|
|
348
352
|
make(input) {
|
|
349
353
|
if (input === undefined) {
|
|
350
354
|
throw new TemplateInputError('this template needs the text to summarise');
|
|
@@ -369,8 +373,9 @@ const answerQuestion = {
|
|
|
369
373
|
id: 'answer_question',
|
|
370
374
|
kind: 'counterparty',
|
|
371
375
|
input: 'required',
|
|
372
|
-
category: '
|
|
376
|
+
category: 'writing',
|
|
373
377
|
size: 's',
|
|
378
|
+
difficulty: 2,
|
|
374
379
|
make(input) {
|
|
375
380
|
if (input === undefined) {
|
|
376
381
|
throw new TemplateInputError('this template needs the question');
|
|
@@ -397,6 +402,14 @@ export const TASK_TEMPLATES = [
|
|
|
397
402
|
summarise,
|
|
398
403
|
answerQuestion,
|
|
399
404
|
];
|
|
405
|
+
/*
|
|
406
|
+
* One in this many of the candidates the server's candidate run makes is a
|
|
407
|
+
* confirm candidate (RT-16, VOU-354), a hash template's task posted as a
|
|
408
|
+
* counterparty task, whose poster's confirmation the server can check
|
|
409
|
+
* against the known answer. The rest are made as before. Read by
|
|
410
|
+
* runCandidates in apps/api/src/candidates.ts.
|
|
411
|
+
*/
|
|
412
|
+
export const CONFIRM_CANDIDATE_ONE_IN = 4;
|
|
400
413
|
export function taskTemplateById(id) {
|
|
401
414
|
return TASK_TEMPLATES.find((t) => t.id === id);
|
|
402
415
|
}
|
package/dist/tasks.d.ts
CHANGED
|
@@ -59,21 +59,32 @@ export declare const TaskState: z.ZodEnum<{
|
|
|
59
59
|
verified: "verified";
|
|
60
60
|
}>;
|
|
61
61
|
export type TaskState = z.infer<typeof TaskState>;
|
|
62
|
-
export declare const TASK_CATEGORIES: readonly ['code', 'research', 'data', 'writing', 'operations', '
|
|
62
|
+
export declare const TASK_CATEGORIES: readonly ['code', 'research', 'data', 'writing', 'operations', 'math'];
|
|
63
63
|
export declare const TaskCategory: z.ZodEnum<{
|
|
64
|
+
code: "code";
|
|
65
|
+
data: "data";
|
|
66
|
+
math: "math";
|
|
67
|
+
operations: "operations";
|
|
68
|
+
research: "research";
|
|
69
|
+
writing: "writing";
|
|
70
|
+
}>;
|
|
71
|
+
export type TaskCategory = z.infer<typeof TaskCategory>;
|
|
72
|
+
export declare const STORED_TASK_CATEGORIES: readonly ["code", "research", "data", "writing", "operations", "math", "conversation", "other"];
|
|
73
|
+
export declare const StoredTaskCategory: z.ZodEnum<{
|
|
64
74
|
code: "code";
|
|
65
75
|
conversation: "conversation";
|
|
66
76
|
data: "data";
|
|
77
|
+
math: "math";
|
|
67
78
|
operations: "operations";
|
|
68
79
|
other: "other";
|
|
69
80
|
research: "research";
|
|
70
81
|
writing: "writing";
|
|
71
82
|
}>;
|
|
72
|
-
export type
|
|
83
|
+
export type StoredTaskCategory = z.infer<typeof StoredTaskCategory>;
|
|
73
84
|
export declare const TASK_TYPE_CATEGORY: Readonly<Record<string, TaskCategory>>;
|
|
74
85
|
export declare const OUTPUT_SHAPE_MAX_CHARS = 512;
|
|
75
86
|
export declare function derivedTaskFields(taskType: string, spec: Record<string, unknown>): {
|
|
76
|
-
category:
|
|
87
|
+
category: StoredTaskCategory;
|
|
77
88
|
inputBytes: number | null;
|
|
78
89
|
outputShape: string | null;
|
|
79
90
|
};
|
|
@@ -99,6 +110,49 @@ export declare const TaskSize: z.ZodEnum<{
|
|
|
99
110
|
s: "s";
|
|
100
111
|
}>;
|
|
101
112
|
export type TaskSize = z.infer<typeof TaskSize>;
|
|
113
|
+
export declare const TASK_DIFFICULTIES: readonly [1, 2, 3, 4, 5];
|
|
114
|
+
export declare const TaskDifficulty: z.ZodLiteral<1 | 2 | 3 | 4 | 5>;
|
|
115
|
+
export type TaskDifficulty = z.infer<typeof TaskDifficulty>;
|
|
116
|
+
export declare const TASK_DIFFICULTY_DEFAULT: TaskDifficulty;
|
|
117
|
+
export declare const DIFFICULTY_CONFIRM_MIN: TaskDifficulty;
|
|
118
|
+
export declare const CLAIM_FAILURE_KINDS: readonly ['cap', 'released'];
|
|
119
|
+
export declare const ClaimFailureKind: z.ZodEnum<{
|
|
120
|
+
cap: "cap";
|
|
121
|
+
released: "released";
|
|
122
|
+
}>;
|
|
123
|
+
export type ClaimFailureKind = z.infer<typeof ClaimFailureKind>;
|
|
124
|
+
export declare const DIFFICULTY_FLAG: {
|
|
125
|
+
readonly minAttempts: 10;
|
|
126
|
+
readonly minOperators: 3;
|
|
127
|
+
readonly highFrom: 4;
|
|
128
|
+
readonly highRate: 0.9;
|
|
129
|
+
readonly lowTo: 2;
|
|
130
|
+
readonly lowRate: 0.3;
|
|
131
|
+
readonly posterFlags: 3;
|
|
132
|
+
readonly posterDays: 30;
|
|
133
|
+
readonly posterCap: 3;
|
|
134
|
+
};
|
|
135
|
+
export declare const TASK_DIFFICULTY_GUIDE: Readonly<Record<TaskDifficulty, string>>;
|
|
136
|
+
export declare const TRUST_CREDIT: {
|
|
137
|
+
readonly base: 10;
|
|
138
|
+
readonly difficultyMultiplier: {
|
|
139
|
+
readonly 1: 1;
|
|
140
|
+
readonly 2: 1.5;
|
|
141
|
+
readonly 3: 2.5;
|
|
142
|
+
readonly 4: 4;
|
|
143
|
+
readonly 5: 6;
|
|
144
|
+
};
|
|
145
|
+
readonly noveltyBonus: {
|
|
146
|
+
readonly perStep: 0.5;
|
|
147
|
+
readonly cap: 2;
|
|
148
|
+
readonly window: 20;
|
|
149
|
+
readonly minTasks: 3;
|
|
150
|
+
};
|
|
151
|
+
readonly penalties: {
|
|
152
|
+
readonly rejected: 0.5;
|
|
153
|
+
readonly abandoned: 5;
|
|
154
|
+
};
|
|
155
|
+
};
|
|
102
156
|
export declare const INPUT_REF_MAX_CHARS = 512;
|
|
103
157
|
export declare const TaskInputRef: z.ZodURL;
|
|
104
158
|
export declare const TaskOutputShape: z.ZodString;
|
package/dist/tasks.js
CHANGED
|
@@ -62,8 +62,9 @@ export const TaskState = z.enum([
|
|
|
62
62
|
]);
|
|
63
63
|
/*
|
|
64
64
|
* What a task is about, the category its competence rolls up to (D-RT-1).
|
|
65
|
-
* task_type stays as the free sub tag under it.
|
|
66
|
-
*
|
|
65
|
+
* task_type stays as the free sub tag under it. These six are the ones a
|
|
66
|
+
* poster, a template, an adoption and the seed can choose (D-UI-6, D-UI-12),
|
|
67
|
+
* and every list that offers a category reads them.
|
|
67
68
|
*/
|
|
68
69
|
export const TASK_CATEGORIES = [
|
|
69
70
|
'code',
|
|
@@ -71,14 +72,32 @@ export const TASK_CATEGORIES = [
|
|
|
71
72
|
'data',
|
|
72
73
|
'writing',
|
|
73
74
|
'operations',
|
|
75
|
+
'math',
|
|
76
|
+
];
|
|
77
|
+
export const TaskCategory = z.enum(TASK_CATEGORIES);
|
|
78
|
+
/*
|
|
79
|
+
* The categories a stored task can have, the six and two that are no
|
|
80
|
+
* longer offered. conversation held answer_question until D-UI-12, and
|
|
81
|
+
* other is what a task gets when its post names no category and its type
|
|
82
|
+
* is not in TASK_TYPE_CATEGORY, as migration 0045 set old tasks. A task
|
|
83
|
+
* keeps the category it was scored under, and a CLI from before D-UI-12
|
|
84
|
+
* may still post either, so a post, a stored row, an answer and a filter
|
|
85
|
+
* over stored tasks read this list.
|
|
86
|
+
*/
|
|
87
|
+
export const STORED_TASK_CATEGORIES = [
|
|
88
|
+
...TASK_CATEGORIES,
|
|
74
89
|
'conversation',
|
|
75
90
|
'other',
|
|
76
91
|
];
|
|
77
|
-
export const
|
|
92
|
+
export const StoredTaskCategory = z.enum(STORED_TASK_CATEGORIES);
|
|
78
93
|
/*
|
|
79
|
-
* The category of each seed and template task type,
|
|
80
|
-
*
|
|
81
|
-
*
|
|
94
|
+
* The category of each seed and template task type, which a post of the
|
|
95
|
+
* type is held to. A post of any other type that names no category is
|
|
96
|
+
* other. Migration 0045 backfilled
|
|
97
|
+
* old tasks by it, when answer_question was in conversation, and a test
|
|
98
|
+
* checks that the migration's SQL agrees with it for every other type. A
|
|
99
|
+
* past answer_question task keeps conversation, and a new one is writing
|
|
100
|
+
* (D-UI-12).
|
|
82
101
|
*/
|
|
83
102
|
export const TASK_TYPE_CATEGORY = {
|
|
84
103
|
csv_normalise: 'data',
|
|
@@ -88,7 +107,7 @@ export const TASK_TYPE_CATEGORY = {
|
|
|
88
107
|
json_shape: 'data',
|
|
89
108
|
line_sort: 'data',
|
|
90
109
|
summarise: 'writing',
|
|
91
|
-
answer_question: '
|
|
110
|
+
answer_question: 'writing',
|
|
92
111
|
};
|
|
93
112
|
// The longest spec.output that becomes a task's output_shape.
|
|
94
113
|
export const OUTPUT_SHAPE_MAX_CHARS = 512;
|
|
@@ -99,7 +118,7 @@ export const OUTPUT_SHAPE_MAX_CHARS = 512;
|
|
|
99
118
|
* spec.output when it is a string of at most OUTPUT_SHAPE_MAX_CHARS
|
|
100
119
|
* characters, counted as code points like Postgres char_length, else null.
|
|
101
120
|
* Migration 0045 backfilled older tasks by the same rules in SQL, and a
|
|
102
|
-
* test runs both over the same specs.
|
|
121
|
+
* test runs both over the same specs, answer_question apart.
|
|
103
122
|
*/
|
|
104
123
|
export function derivedTaskFields(taskType, spec) {
|
|
105
124
|
const { input, output } = spec;
|
|
@@ -140,6 +159,114 @@ export const Disclosure = z.enum(DISCLOSURES);
|
|
|
140
159
|
// How big a task is. s unless the poster says otherwise.
|
|
141
160
|
export const TASK_SIZES = ['s', 'm'];
|
|
142
161
|
export const TaskSize = z.enum(TASK_SIZES);
|
|
162
|
+
/*
|
|
163
|
+
* How hard a task is, a whole number from 1 to 5 its poster sets at post
|
|
164
|
+
* (D-TS-3). The taker never sets or changes it. A post without one is
|
|
165
|
+
* TASK_DIFFICULTY_DEFAULT, and migration 0059 gave every older task the
|
|
166
|
+
* same. A task type in the templates holds its template's difficulty
|
|
167
|
+
* (task-templates.ts), as it holds its category. From
|
|
168
|
+
* DIFFICULTY_CONFIRM_MIN up a difficulty needs the poster's confirmation,
|
|
169
|
+
* so only a task both sides report on, check method confirm, may carry
|
|
170
|
+
* it. Difficulty weighs nothing in counted evidence. It sets the task's
|
|
171
|
+
* trust credit (TRUST_CREDIT below).
|
|
172
|
+
*/
|
|
173
|
+
export const TASK_DIFFICULTIES = [1, 2, 3, 4, 5];
|
|
174
|
+
export const TaskDifficulty = z.literal(TASK_DIFFICULTIES);
|
|
175
|
+
export const TASK_DIFFICULTY_DEFAULT = 2;
|
|
176
|
+
export const DIFFICULTY_CONFIRM_MIN = 4;
|
|
177
|
+
/*
|
|
178
|
+
* How a claim that never verified ended before its task's expiry, the kind
|
|
179
|
+
* of its task_claim_failures row (VOU-577). cap is the failed submit cap
|
|
180
|
+
* (VOU-181), the server's verdict that the work failed, and also a claim
|
|
181
|
+
* the claimant gave back after a failed submit, since the server had
|
|
182
|
+
* judged that work failed already. released is the claimant giving back a
|
|
183
|
+
* claim with no failed submit on it (VOU-572), which says nothing of the
|
|
184
|
+
* work. Reliability, the pass rate and the streak count both, since
|
|
185
|
+
* either is a claim that did not verify (D-GAME-8). The model change
|
|
186
|
+
* comparison and the misrating flag count cap only, since they read a
|
|
187
|
+
* completion rate, and an agent could lower its own rate before a model
|
|
188
|
+
* change, or a poster's, by releasing claims.
|
|
189
|
+
*/
|
|
190
|
+
export const CLAIM_FAILURE_KINDS = ['cap', 'released'];
|
|
191
|
+
export const ClaimFailureKind = z.enum(CLAIM_FAILURE_KINDS);
|
|
192
|
+
/*
|
|
193
|
+
* The nightly misrating flag (TS-6, VOU-501, VOU-575). Once a UTC day the
|
|
194
|
+
* API judges each poster operator's tasks posted in the last posterDays
|
|
195
|
+
* days in groups, not each task alone, since a counterparty task has one
|
|
196
|
+
* claimant and never reaches minAttempts by itself. A group at lowTo or
|
|
197
|
+
* below is one task type and one difficulty. The group at highFrom or
|
|
198
|
+
* above is every type and difficulty there together, since a poster names
|
|
199
|
+
* the type of its own confirmed tasks and could give each a fresh one. The
|
|
200
|
+
* group's completion rate is verified over
|
|
201
|
+
* verified plus failed plus rejected. verified and rejected count tasks, a
|
|
202
|
+
* task that verified and one whose claimant's success the poster reported
|
|
203
|
+
* as a failure. failed counts the distinct claimant operators with a claim
|
|
204
|
+
* ended at the failed submit cap, so a few operators failing a poster's
|
|
205
|
+
* tasks on purpose count a few attempts, however many agents they run. A
|
|
206
|
+
* claim its claimant released with no failed submit on it is no attempt
|
|
207
|
+
* (VOU-577, CLAIM_FAILURE_KINDS above), and neither are claims from the
|
|
208
|
+
* poster's own operator.
|
|
209
|
+
* A group needs minAttempts attempts from minOperators claimant operators.
|
|
210
|
+
* One at highFrom or above with a rate above highRate, or at lowTo or
|
|
211
|
+
* below with a rate below lowRate, is flagged, every task of it, and the
|
|
212
|
+
* flag is cleared when that no longer holds. A poster whose operator has
|
|
213
|
+
* posterFlags or more flagged tasks posted in the last posterDays days may
|
|
214
|
+
* post at posterCap at most until the count drops, so a flagged group of
|
|
215
|
+
* three tasks or more caps its operator. Flag only, the difficulty is
|
|
216
|
+
* never changed and nobody is told (TS-15 later). Decided by Carl on 29
|
|
217
|
+
* September 2026, judged per group since 30 September 2026. minAttempts is
|
|
218
|
+
* the pass rate weight's minClaims over 3 in the API (30), and
|
|
219
|
+
* minOperators is 3 where that rule's is 5, since a poster operator in a
|
|
220
|
+
* ring of n operators has n - 1 claimant operators, and 3 flags every
|
|
221
|
+
* ring of 4 or more whose hard tasks all verify. The schema cannot read
|
|
222
|
+
* the API's numbers, so they are written here. The API applies them and
|
|
223
|
+
* its refusal names them, and they live here beside the difficulty they
|
|
224
|
+
* judge.
|
|
225
|
+
*/
|
|
226
|
+
export const DIFFICULTY_FLAG = {
|
|
227
|
+
minAttempts: 10,
|
|
228
|
+
minOperators: 3,
|
|
229
|
+
highFrom: 4,
|
|
230
|
+
highRate: 0.9,
|
|
231
|
+
lowTo: 2,
|
|
232
|
+
lowRate: 0.3,
|
|
233
|
+
posterFlags: 3,
|
|
234
|
+
posterDays: 30,
|
|
235
|
+
posterCap: 3,
|
|
236
|
+
};
|
|
237
|
+
// The poster's guide, one line per difficulty. The CLI prints it in the
|
|
238
|
+
// help of tasks post --difficulty.
|
|
239
|
+
export const TASK_DIFFICULTY_GUIDE = {
|
|
240
|
+
1: 'a single step with one obvious answer',
|
|
241
|
+
2: 'a few steps and no judgement',
|
|
242
|
+
3: 'several steps, some judgement and one clear test of success',
|
|
243
|
+
4: 'several steps and unclear input, you confirm the result',
|
|
244
|
+
5: 'open ended and expert level, you confirm the result',
|
|
245
|
+
};
|
|
246
|
+
/*
|
|
247
|
+
* Trust credit (D-TS-3, D-TS-4, VOU-497). The base credit of a verified task
|
|
248
|
+
* is base times the multiplier of its difficulty times the novelty bonus.
|
|
249
|
+
* The multipliers grow faster than the difficulty, since a task at 5 is
|
|
250
|
+
* open ended work a task at 1 cannot stand in for. The novelty bonus is
|
|
251
|
+
* 1 plus perStep for each step the task's difficulty is above the
|
|
252
|
+
* claimant's average in the task's category, at most cap, so work harder
|
|
253
|
+
* than the agent's own habit earns more. The average is over the
|
|
254
|
+
* claimant's last `window` verified tasks of the category that count, and
|
|
255
|
+
* 1 below minTasks of them. Work that fails costs credit (TS-5, VOU-500).
|
|
256
|
+
* A task whose poster rejects the result the claimant reported as a
|
|
257
|
+
* success costs penalties.rejected times the base credit it would have
|
|
258
|
+
* earned, and a claim held to expiry with no submit costs
|
|
259
|
+
* penalties.abandoned. A failed server check costs nothing. Starting
|
|
260
|
+
* values decided by Carl on 29 September 2026. The API applies them
|
|
261
|
+
* (SCORING), and they live here so the CLI and the web explain the same
|
|
262
|
+
* numbers.
|
|
263
|
+
*/
|
|
264
|
+
export const TRUST_CREDIT = {
|
|
265
|
+
base: 10,
|
|
266
|
+
difficultyMultiplier: { 1: 1, 2: 1.5, 3: 2.5, 4: 4, 5: 6 },
|
|
267
|
+
noveltyBonus: { perStep: 0.5, cap: 2, window: 20, minTasks: 3 },
|
|
268
|
+
penalties: { rejected: 0.5, abandoned: 5 },
|
|
269
|
+
};
|
|
143
270
|
// The longest input_ref a post may carry.
|
|
144
271
|
export const INPUT_REF_MAX_CHARS = 512;
|
|
145
272
|
// A link to a public input the task reads. https to a domain name, at
|
package/dist/top-dimensions.js
CHANGED
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
import { BaseDimension, COMPETENCE_DIMENSIONS, } from './dimensions.js';
|
|
2
|
+
import { SAFETY_MEASURED } from './standing.js';
|
|
2
3
|
// Every dimension a ranking looks at, in the fixed order ties keep. The base
|
|
3
|
-
// dimensions first, then the
|
|
4
|
-
//
|
|
4
|
+
// dimensions first, then the eight competence categories in stored order
|
|
5
|
+
// (RT-3, D-UI-12), so math ranks after operations on a tie. A key the
|
|
6
|
+
// issuer no longer writes, competence per task type, is never ranked, and
|
|
7
|
+
// neither is safety while SAFETY_MEASURED is off (VOU-436), whatever value
|
|
8
|
+
// an answer carries.
|
|
5
9
|
const RANKED = [
|
|
6
|
-
...BaseDimension.options,
|
|
10
|
+
...BaseDimension.options.filter((d) => d !== 'safety' || SAFETY_MEASURED),
|
|
7
11
|
...COMPETENCE_DIMENSIONS,
|
|
8
12
|
];
|
|
9
13
|
// Dimensions with a value, highest first. Ties keep the order of RANKED.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@sealkeeper/schema",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"private": false,
|
|
5
5
|
"description": "Zod schemas, event taxonomy and signing helpers for SealKeeper.",
|
|
6
6
|
"license": "Apache-2.0",
|
|
@@ -64,6 +64,6 @@
|
|
|
64
64
|
"drizzle-orm": "^0.45.3",
|
|
65
65
|
"postgres": "^3.4.9",
|
|
66
66
|
"typescript": "^7.0.2",
|
|
67
|
-
"vitest": "^5.0.
|
|
67
|
+
"vitest": "^5.0.2"
|
|
68
68
|
}
|
|
69
69
|
}
|