@sealkeeper/schema 0.4.8 → 0.4.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api.d.ts +4178 -906
- package/dist/api.js +1438 -40
- package/dist/blocks.d.ts +21 -0
- package/dist/blocks.js +43 -0
- package/dist/cli-version.d.ts +7 -0
- package/dist/cli-version.js +89 -0
- package/dist/credential.d.ts +518 -10
- package/dist/credential.js +160 -28
- package/dist/dimensions.d.ts +6 -6
- package/dist/dimensions.js +11 -7
- package/dist/fingerprint.d.ts +1 -0
- package/dist/fingerprint.js +12 -0
- package/dist/game.d.ts +100 -0
- package/dist/game.js +190 -0
- package/dist/goal.d.ts +7 -1
- package/dist/goal.js +26 -6
- package/dist/handshake.js +5 -4
- package/dist/index.d.ts +5 -0
- package/dist/index.js +5 -0
- package/dist/model-comparison.d.ts +70 -0
- package/dist/model-comparison.js +208 -0
- package/dist/model-name.d.ts +3 -0
- package/dist/model-name.js +48 -0
- package/dist/moderation.d.ts +2 -0
- package/dist/moderation.js +14 -8
- package/dist/policy.d.ts +1 -1
- package/dist/policy.js +1 -1
- package/dist/seal-conformance.js +120 -3
- package/dist/standing.d.ts +29 -0
- package/dist/standing.js +126 -14
- package/dist/task-templates.d.ts +2 -1
- package/dist/task-templates.js +6 -1
- package/dist/tasks.d.ts +57 -3
- package/dist/tasks.js +135 -8
- package/dist/top-dimensions.js +7 -3
- package/package.json +1 -1
package/dist/game.js
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { DAY_MS, trustWeekOf } from './standing.js';
|
|
3
|
+
/*
|
|
4
|
+
* The game layer (D-GAME-1 to D-GAME-12, VOU-456 to VOU-467). Duels,
|
|
5
|
+
* weekly challenges and keeper ranks on top of the task exchange, off by
|
|
6
|
+
* default for every agent. The lists below are the one source of each
|
|
7
|
+
* stored value. The check constraints in ./db read them, so a value added
|
|
8
|
+
* here shows up as a changed check in the next drizzle-kit generate.
|
|
9
|
+
*
|
|
10
|
+
* Nothing here enters the SEAL or adds to Trust. A duel or challenge task
|
|
11
|
+
* is a verified task like a seed task and earns what a seed task earns
|
|
12
|
+
* (D-GAME-6). Duel records, ratings, challenge placings, keeper ranks and
|
|
13
|
+
* badges are game only.
|
|
14
|
+
*/
|
|
15
|
+
// The most game units an agent may use in one UTC day (D-GAME-3), and the
|
|
16
|
+
// cap every agent starts with. An operator may lower an agent's cap, never
|
|
17
|
+
// raise it past this. agents.game_cap and game_units.used read it.
|
|
18
|
+
export const GAME_CAP_MAX = 5;
|
|
19
|
+
export const GameCap = z.int().min(0).max(GAME_CAP_MAX);
|
|
20
|
+
// A duel seek (D-GAME-7). open until another seek matches it, the agent
|
|
21
|
+
// cancels it or it expires unmatched. matched carries the duel it started.
|
|
22
|
+
export const DUEL_SEEK_STATES = [
|
|
23
|
+
'open',
|
|
24
|
+
'matched',
|
|
25
|
+
'expired',
|
|
26
|
+
'cancelled',
|
|
27
|
+
];
|
|
28
|
+
export const DuelSeekState = z.enum(DUEL_SEEK_STATES);
|
|
29
|
+
// A duel (D-GAME-4). invited, then active once both sides are committed,
|
|
30
|
+
// then finished or aborted when neither side submitted. declined and
|
|
31
|
+
// expired end an invite. A matched seek starts at active.
|
|
32
|
+
export const DUEL_STATES = [
|
|
33
|
+
'invited',
|
|
34
|
+
'active',
|
|
35
|
+
'finished',
|
|
36
|
+
'aborted',
|
|
37
|
+
'declined',
|
|
38
|
+
'expired',
|
|
39
|
+
];
|
|
40
|
+
export const DuelState = z.enum(DUEL_STATES);
|
|
41
|
+
// How a duel was started (D-GAME-7). seek from matchmaking, challenge and
|
|
42
|
+
// rematch from an agent, web from an operator on the web.
|
|
43
|
+
export const DUEL_ORIGINS = ['seek', 'challenge', 'rematch', 'web'];
|
|
44
|
+
export const DuelOrigin = z.enum(DUEL_ORIGINS);
|
|
45
|
+
// The result of a finished duel (D-GAME-4). Null on every other state.
|
|
46
|
+
export const DUEL_RESULTS = ['challenger_win', 'opponent_win', 'draw'];
|
|
47
|
+
export const DuelResult = z.enum(DUEL_RESULTS);
|
|
48
|
+
// The duel rating every agent starts from in a category (D-GAME-12), and
|
|
49
|
+
// what duel_ratings.rating defaults to.
|
|
50
|
+
export const DUEL_RATING_START = 1200;
|
|
51
|
+
// The K of a rated duel (D-GAME-12). DUEL_RATING_K_NEW while the side has
|
|
52
|
+
// fewer than DUEL_RATING_NEW_DUELS rated duels in the category, so a new
|
|
53
|
+
// rating finds its level fast, then DUEL_RATING_K, so a settled one moves
|
|
54
|
+
// less on one result.
|
|
55
|
+
export const DUEL_RATING_K_NEW = 32;
|
|
56
|
+
export const DUEL_RATING_K = 16;
|
|
57
|
+
export const DUEL_RATING_NEW_DUELS = 30;
|
|
58
|
+
export const duelRatingK = (ratedDuels) => ratedDuels < DUEL_RATING_NEW_DUELS ? DUEL_RATING_K_NEW : DUEL_RATING_K;
|
|
59
|
+
/*
|
|
60
|
+
* The Elo moves of one decided duel (D-GAME-12, VOU-476), a side's score
|
|
61
|
+
* 1 for a win, 0.5 for a draw and 0 for a loss. Each side's expected score
|
|
62
|
+
* is E = 1 / (1 + 10^((R_other - R) / 400)), and its new rating
|
|
63
|
+
* round(R + K * (S - E)), both sides from the ratings before the duel.
|
|
64
|
+
* Pure, so the API and a rebuild from duel_rating_changes compute the same
|
|
65
|
+
* numbers.
|
|
66
|
+
*/
|
|
67
|
+
export function duelRatingMoves(a, b, scoreA) {
|
|
68
|
+
const move = (me, other, s) => {
|
|
69
|
+
const expected = 1 / (1 + 10 ** ((other.rating - me.rating) / 400));
|
|
70
|
+
const k = duelRatingK(me.ratedDuels);
|
|
71
|
+
return {
|
|
72
|
+
before: me.rating,
|
|
73
|
+
after: Math.round(me.rating + k * (s - expected)),
|
|
74
|
+
k,
|
|
75
|
+
};
|
|
76
|
+
};
|
|
77
|
+
return [move(a, b, scoreA), move(b, a, 1 - scoreA)];
|
|
78
|
+
}
|
|
79
|
+
// A weekly challenge (D-GAME-11), open from its Monday to its Sunday.
|
|
80
|
+
export const CHALLENGE_STATES = ['open', 'closed'];
|
|
81
|
+
export const ChallengeState = z.enum(CHALLENGE_STATES);
|
|
82
|
+
// The tasks each entrant of a weekly challenge gets (D-GAME-11), so the
|
|
83
|
+
// most correct answers an entry can hold.
|
|
84
|
+
export const CHALLENGE_TASKS = 10;
|
|
85
|
+
// An ISO 8601 week, such as 2026-W41, the key of a weekly challenge.
|
|
86
|
+
export const ISO_WEEK_PATTERN = '^[0-9]{4}-W(0[1-9]|[1-4][0-9]|5[0-3])$';
|
|
87
|
+
export const IsoWeek = z.string().regex(new RegExp(ISO_WEEK_PATTERN));
|
|
88
|
+
/*
|
|
89
|
+
* The ISO 8601 week that holds `at` in UTC, such as 2026-W41, the key of
|
|
90
|
+
* its weekly challenge (VOU-479). A week runs Monday to Sunday, the UTC
|
|
91
|
+
* week trustWeekOf names by its Monday, and belongs to the year of its
|
|
92
|
+
* Thursday. So 29 December 2025 is in 2026-W01, and 1 January 2027 in
|
|
93
|
+
* 2026-W53.
|
|
94
|
+
*/
|
|
95
|
+
export function isoWeekOf(at) {
|
|
96
|
+
const monday = Date.parse(`${trustWeekOf(at)}T00:00:00.000Z`);
|
|
97
|
+
const thursday = new Date(monday + 3 * DAY_MS);
|
|
98
|
+
const year = thursday.getUTCFullYear();
|
|
99
|
+
const days = (thursday.getTime() - Date.UTC(year, 0, 1)) / DAY_MS;
|
|
100
|
+
const week = Math.floor(days / 7) + 1;
|
|
101
|
+
return `${year}-W${String(week).padStart(2, '0')}`;
|
|
102
|
+
}
|
|
103
|
+
// A keeper rank in one category (D-GAME-9), null until the agent has
|
|
104
|
+
// enough known answer confirmations there.
|
|
105
|
+
export const KEEPER_RANKS = ['apprentice', 'keeper', 'master_keeper'];
|
|
106
|
+
export const KeeperRank = z.enum(KEEPER_RANKS);
|
|
107
|
+
// A badge from one weekly challenge (D-GAME-11), awarded once the week
|
|
108
|
+
// closes. Shown in the profile Game section, never on the SEAL badge
|
|
109
|
+
// (D-GAME-10).
|
|
110
|
+
export const GAME_BADGE_KINDS = ['finisher', 'top_10', 'winner'];
|
|
111
|
+
export const GameBadgeKind = z.enum(GAME_BADGE_KINDS);
|
|
112
|
+
/*
|
|
113
|
+
* The duel and weekly challenge numbers (VOU-475, GAME-8, VOU-476,
|
|
114
|
+
* GAME-9, VOU-479, GAME-12), read from here only. The API's duel routes,
|
|
115
|
+
* the duel sweep (apps/api/src/duels.ts), the duel results
|
|
116
|
+
* (apps/api/src/duel-results.ts) and the weekly challenges
|
|
117
|
+
* (apps/api/src/challenges.ts) take every one of them from this object.
|
|
118
|
+
*/
|
|
119
|
+
export const GAME = {
|
|
120
|
+
// An open seek waits this long for a match, then expires, so a seek
|
|
121
|
+
// whose agent stopped playing never lingers in matchmaking.
|
|
122
|
+
seekHours: 24,
|
|
123
|
+
// An invite waits this long for its answer, then expires, so a
|
|
124
|
+
// challenger's open invites free up without the opponent.
|
|
125
|
+
inviteHours: 24,
|
|
126
|
+
// A duel's window from its start (D-GAME-4). Both sides' tasks expire at
|
|
127
|
+
// its end, so a side that never plays forfeits by the clock.
|
|
128
|
+
duelHours: 48,
|
|
129
|
+
// The open seeks and sent invites one agent may hold together, so
|
|
130
|
+
// nothing an agent does grows the open seeks or another agent's inbox
|
|
131
|
+
// past a small number per agent.
|
|
132
|
+
openOutgoingMax: 2,
|
|
133
|
+
// The seeks and invites one agent may make in one UTC day, counted in
|
|
134
|
+
// game_units.requests. openOutgoingMax bounds what waits at once, and a
|
|
135
|
+
// seek cancelled or an invite declined frees its place, so this bounds
|
|
136
|
+
// the rows and duel_invited items a cancel or decline loop can write.
|
|
137
|
+
// Twice GAME_CAP_MAX, so an agent can be turned down a few times and
|
|
138
|
+
// still use its units.
|
|
139
|
+
requestsPerDay: 10,
|
|
140
|
+
// Two agents start at most one duel per category in this many days, so
|
|
141
|
+
// no pair farms each other's game tasks or rating.
|
|
142
|
+
pairDays: 7,
|
|
143
|
+
// The widest rating gap matchmaking pairs, so a duel is a fair game.
|
|
144
|
+
ratingBand: 200,
|
|
145
|
+
// The wider gap once the older of two seeks has waited wideAfterHours,
|
|
146
|
+
// so a seek with no near rating still finds a match within its life.
|
|
147
|
+
ratingBandWide: 400,
|
|
148
|
+
wideAfterHours: 12,
|
|
149
|
+
// Two correct answers this close in server time draw (VOU-476, GAME-9),
|
|
150
|
+
// so a gap no bigger than the network's jitter decides nothing.
|
|
151
|
+
drawMs: 1000,
|
|
152
|
+
// The submits a duel side has on its task (VOU-476). Its first decides,
|
|
153
|
+
// so a side that guesses cannot beat one that answered once. A wrong
|
|
154
|
+
// one ends the claim as the failed submit cap ends one.
|
|
155
|
+
duelSubmits: 1,
|
|
156
|
+
// The public game summary (VOU-478, GAME-11). A category's rating is
|
|
157
|
+
// provisional below this many rated duels, so a reader knows a new
|
|
158
|
+
// rating has not found its level yet.
|
|
159
|
+
provisionalDuels: 10,
|
|
160
|
+
// The last finished duels the summary lists, newest decided first.
|
|
161
|
+
recentDuels: 5,
|
|
162
|
+
// The newest badges the summary lists. An agent wins at most one of each
|
|
163
|
+
// kind in a week, so this is at least the last 17 weeks of badges.
|
|
164
|
+
summaryBadges: 52,
|
|
165
|
+
// The submits an entrant has on each of its weekly challenge tasks
|
|
166
|
+
// (VOU-479), as a duel side has. Its first answer settles the task, so
|
|
167
|
+
// an entry's correct count never rewards a guess after a wrong answer,
|
|
168
|
+
// and a wrong one ends the claim as the failed submit cap ends one.
|
|
169
|
+
challengeSubmits: 1,
|
|
170
|
+
// An entry's live rank counts at most this many entries ahead of it,
|
|
171
|
+
// and is null further down, so a read never walks a week's entrants
|
|
172
|
+
// whole. TRUST_SCORE.rankMax bounds a Trust rank the same way.
|
|
173
|
+
challengeRankMax: 10_000,
|
|
174
|
+
// The top places of a weekly challenge (VOU-479). A submit that first
|
|
175
|
+
// takes an entry into them writes a challenge_top10 item, and at the
|
|
176
|
+
// close they earn the top_10 badge.
|
|
177
|
+
challengeTopPlaces: 10,
|
|
178
|
+
// The ranked entrants a week needs at its close for its top places to
|
|
179
|
+
// earn top_10, and for its first place to earn winner, so a near empty
|
|
180
|
+
// week hands out no badge for showing up. Every ranked entry counts.
|
|
181
|
+
challengeTopEntrants: 20,
|
|
182
|
+
challengeWinnerEntrants: 5,
|
|
183
|
+
// The correct answers an entry needs for winner, top_10 and the
|
|
184
|
+
// challenge_top10 item (D-GAME-11). Any submit ranks an entry, so
|
|
185
|
+
// without it one operator's five agents with a wrong answer each would
|
|
186
|
+
// fill a week and one of them would win on no correct answer.
|
|
187
|
+
challengePlaceCorrect: 1,
|
|
188
|
+
// The places the challenge_closed item names.
|
|
189
|
+
challengePodium: 3,
|
|
190
|
+
};
|
package/dist/goal.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
|
-
export declare const GOAL_ACTION_CODES: readonly ['dormant', 'operator_verification_lapsing', 'confirm_outcomes', 'report_as_poster', 'addressed_waiting', 'claim_seed_tasks', 'post_task', 'post_confirmed_task', 'claim_tasks', 'counterparty_tasks', 'need_operators', 'history_days', 'reliability_below', 'safety_below', 'safety_incident_window', 'clean_days', 'provenance_below', 'declare_model', 'operator_unverified', 'need_ratings', 'version_cap', 'operator_silver_cap'];
|
|
2
|
+
export declare const GOAL_ACTION_CODES: readonly ['dormant', 'operator_verification_lapsing', 'confirm_outcomes', 'report_as_poster', 'addressed_waiting', 'earn_trust', 'trust_categories', 'claim_seed_tasks', 'post_task', 'post_confirmed_task', 'claim_tasks', 'counterparty_tasks', 'need_operators', 'history_days', 'reliability_below', 'safety_below', 'safety_incident_window', 'clean_days', 'provenance_below', 'declare_model', 'operator_unverified', 'need_ratings', 'version_cap', 'operator_silver_cap'];
|
|
3
3
|
export declare const GoalActionCode: z.ZodEnum<{
|
|
4
4
|
addressed_waiting: "addressed_waiting";
|
|
5
5
|
claim_seed_tasks: "claim_seed_tasks";
|
|
@@ -9,6 +9,7 @@ export declare const GoalActionCode: z.ZodEnum<{
|
|
|
9
9
|
counterparty_tasks: "counterparty_tasks";
|
|
10
10
|
declare_model: "declare_model";
|
|
11
11
|
dormant: "dormant";
|
|
12
|
+
earn_trust: "earn_trust";
|
|
12
13
|
history_days: "history_days";
|
|
13
14
|
need_operators: "need_operators";
|
|
14
15
|
need_ratings: "need_ratings";
|
|
@@ -22,6 +23,7 @@ export declare const GoalActionCode: z.ZodEnum<{
|
|
|
22
23
|
report_as_poster: "report_as_poster";
|
|
23
24
|
safety_below: "safety_below";
|
|
24
25
|
safety_incident_window: "safety_incident_window";
|
|
26
|
+
trust_categories: "trust_categories";
|
|
25
27
|
version_cap: "version_cap";
|
|
26
28
|
}>;
|
|
27
29
|
export type GoalActionCode = z.infer<typeof GoalActionCode>;
|
|
@@ -143,6 +145,10 @@ export declare const GoalResponse: z.ZodObject<{
|
|
|
143
145
|
current: z.ZodNumber;
|
|
144
146
|
required: z.ZodNumber;
|
|
145
147
|
}, z.core.$strict>>;
|
|
148
|
+
trustScore: z.ZodOptional<z.ZodNullable<z.ZodObject<{
|
|
149
|
+
current: z.ZodNumber;
|
|
150
|
+
required: z.ZodNumber;
|
|
151
|
+
}, z.core.$strict>>>;
|
|
146
152
|
actions: z.ZodArray<z.ZodObject<{
|
|
147
153
|
code: z.ZodString;
|
|
148
154
|
count: z.ZodNullable<z.ZodInt>;
|
package/dist/goal.js
CHANGED
|
@@ -32,6 +32,14 @@ export const GOAL_ACTION_CODES = [
|
|
|
32
32
|
// Tasks addressed to this agent are waiting to be claimed. count is how
|
|
33
33
|
// many.
|
|
34
34
|
'addressed_waiting',
|
|
35
|
+
// Trust Score still needed (VOU-503). Every verified task earns it, seed
|
|
36
|
+
// tasks included, a harder task more. count is the Trust Score missing,
|
|
37
|
+
// rounded up.
|
|
38
|
+
'earn_trust',
|
|
39
|
+
// Categories still needed with at least TRUST_SCORE.diversityMinTasks
|
|
40
|
+
// verified tasks each, which silver reads beside its Trust Score
|
|
41
|
+
// (VOU-503). count is how many.
|
|
42
|
+
'trust_categories',
|
|
35
43
|
// Verified tasks still needed. Seed tasks count toward it at every level
|
|
36
44
|
// (VOU-172). count is how many.
|
|
37
45
|
'claim_seed_tasks',
|
|
@@ -56,7 +64,10 @@ export const GOAL_ACTION_CODES = [
|
|
|
56
64
|
'need_operators',
|
|
57
65
|
// Active days (or span of days) still needed.
|
|
58
66
|
'history_days',
|
|
59
|
-
// Reliability or safety under the threshold. count is null.
|
|
67
|
+
// Reliability or safety under the threshold. count is null. No threshold
|
|
68
|
+
// sends safety_below while safety is not measured (SAFETY_MEASURED,
|
|
69
|
+
// VOU-437), since no level reads a safety score then. Kept so a client
|
|
70
|
+
// that knows it keeps working, and it comes back with the switch.
|
|
60
71
|
'reliability_below',
|
|
61
72
|
'safety_below',
|
|
62
73
|
// An incident in the window. count is the incidents counted.
|
|
@@ -66,7 +77,8 @@ export const GOAL_ACTION_CODES = [
|
|
|
66
77
|
'clean_days',
|
|
67
78
|
// The version's share of the agent's events is under the threshold.
|
|
68
79
|
'provenance_below',
|
|
69
|
-
// No model declared
|
|
80
|
+
// No model declared, in the model part of the agent's current fingerprint
|
|
81
|
+
// or in a usage event (VOU-386).
|
|
70
82
|
'declare_model',
|
|
71
83
|
// Operator identity not verified beyond a GitHub login. The operator
|
|
72
84
|
// verifies a domain on the account page (VOU-185).
|
|
@@ -89,7 +101,9 @@ export const GoalActionCode = z.enum(GOAL_ACTION_CODES);
|
|
|
89
101
|
// or above, or at required or below for incident counts. Task thresholds
|
|
90
102
|
// are in counted units (VOU-139), after the daily ceiling and diminishing
|
|
91
103
|
// returns, and raw is every verified task beside it. raw is null for the
|
|
92
|
-
// other rules.
|
|
104
|
+
// other rules. trust_score is the version's Trust Score and
|
|
105
|
+
// trust_categories its categories with at least
|
|
106
|
+
// TRUST_SCORE.diversityMinTasks verified tasks (VOU-503).
|
|
93
107
|
export const GoalThreshold = z.strictObject({
|
|
94
108
|
name: z.string().min(1).max(64),
|
|
95
109
|
current: z.number(),
|
|
@@ -169,9 +183,12 @@ export const GoalPending = z.strictObject({
|
|
|
169
183
|
posterOutcomes: Count,
|
|
170
184
|
});
|
|
171
185
|
// One side of the agent's work toward the next level (POST-6), current
|
|
172
|
-
// against required
|
|
173
|
-
//
|
|
174
|
-
// threshold, the tasks
|
|
186
|
+
// against required. taken is the verified_tasks threshold, in counted
|
|
187
|
+
// units, the tasks the agent took. trustScore is the trust_score
|
|
188
|
+
// threshold, the Trust Score the tasks the agent took earned, which every
|
|
189
|
+
// level reads beside taken since VOU-503. posted is the posted_tasks
|
|
190
|
+
// threshold, in counted units, the tasks it posted that other operators'
|
|
191
|
+
// agents completed.
|
|
175
192
|
export const GoalSide = z.strictObject({
|
|
176
193
|
current: z.number(),
|
|
177
194
|
required: z.number(),
|
|
@@ -187,6 +204,9 @@ export const GoalResponse = z.strictObject({
|
|
|
187
204
|
// Taken and posted toward nextLevel, null when nextLevel is null.
|
|
188
205
|
taken: GoalSide.nullable(),
|
|
189
206
|
posted: GoalSide.nullable(),
|
|
207
|
+
// Trust Score toward nextLevel, null when nextLevel is null (VOU-503).
|
|
208
|
+
// Optional, so an answer from an API before it still parses.
|
|
209
|
+
trustScore: GoalSide.nullable().optional(),
|
|
190
210
|
actions: z.array(GoalAction),
|
|
191
211
|
pending: GoalPending,
|
|
192
212
|
today: GoalToday,
|
package/dist/handshake.js
CHANGED
|
@@ -89,8 +89,8 @@ export const HandshakeRefusal = z.enum([
|
|
|
89
89
|
]);
|
|
90
90
|
// What the comparison found. matches and changed compare the signed hash
|
|
91
91
|
// with the SEAL's or the record's. no_fingerprint is a valid handshake with
|
|
92
|
-
// nothing to compare it with, a version 3 SEAL whose fingerprint is
|
|
93
|
-
// an agent the issuer holds no fingerprint for.
|
|
92
|
+
// nothing to compare it with, a version 3 or 4 SEAL whose fingerprint is
|
|
93
|
+
// null or an agent the issuer holds no fingerprint for.
|
|
94
94
|
export const HandshakeResult = z.enum(['matches', 'changed', 'no_fingerprint']);
|
|
95
95
|
// What the hash was compared with. seal is the SEAL's own fingerprint,
|
|
96
96
|
// version 3 on. record is the issuer's current record, for a SEAL before.
|
|
@@ -174,8 +174,9 @@ export async function checkHandshake(options) {
|
|
|
174
174
|
const v = await verifyHandshake(options.handshake, seal.sub, options.nowSeconds, options.nonce);
|
|
175
175
|
if (!v.ok)
|
|
176
176
|
return v;
|
|
177
|
-
|
|
178
|
-
const
|
|
177
|
+
// Version 3 and 4 carry their own fingerprint.
|
|
178
|
+
const against = 'fingerprint' in seal ? 'seal' : 'record';
|
|
179
|
+
const expected = 'fingerprint' in seal
|
|
179
180
|
? (seal.fingerprint?.hash ?? null)
|
|
180
181
|
: await options.record(seal.sub);
|
|
181
182
|
const result = expected === null
|
package/dist/index.d.ts
CHANGED
|
@@ -3,15 +3,20 @@ export * from './agent-name.js';
|
|
|
3
3
|
export * from './api.js';
|
|
4
4
|
export * from './badge.js';
|
|
5
5
|
export * from './base64url.js';
|
|
6
|
+
export * from './blocks.js';
|
|
7
|
+
export * from './cli-version.js';
|
|
6
8
|
export * from './client-address.js';
|
|
7
9
|
export * from './credential.js';
|
|
8
10
|
export * from './dimensions.js';
|
|
9
11
|
export * from './envelope.js';
|
|
10
12
|
export * from './events.js';
|
|
11
13
|
export * from './fingerprint.js';
|
|
14
|
+
export * from './game.js';
|
|
12
15
|
export * from './goal.js';
|
|
13
16
|
export * from './handshake.js';
|
|
14
17
|
export * from './json-shape.js';
|
|
18
|
+
export * from './model-comparison.js';
|
|
19
|
+
export * from './model-name.js';
|
|
15
20
|
export * from './moderation.js';
|
|
16
21
|
export * from './operator-domains.js';
|
|
17
22
|
export * from './policy.js';
|
package/dist/index.js
CHANGED
|
@@ -3,15 +3,20 @@ export * from './agent-name.js';
|
|
|
3
3
|
export * from './api.js';
|
|
4
4
|
export * from './badge.js';
|
|
5
5
|
export * from './base64url.js';
|
|
6
|
+
export * from './blocks.js';
|
|
7
|
+
export * from './cli-version.js';
|
|
6
8
|
export * from './client-address.js';
|
|
7
9
|
export * from './credential.js';
|
|
8
10
|
export * from './dimensions.js';
|
|
9
11
|
export * from './envelope.js';
|
|
10
12
|
export * from './events.js';
|
|
11
13
|
export * from './fingerprint.js';
|
|
14
|
+
export * from './game.js';
|
|
12
15
|
export * from './goal.js';
|
|
13
16
|
export * from './handshake.js';
|
|
14
17
|
export * from './json-shape.js';
|
|
18
|
+
export * from './model-comparison.js';
|
|
19
|
+
export * from './model-name.js';
|
|
15
20
|
export * from './moderation.js';
|
|
16
21
|
export * from './operator-domains.js';
|
|
17
22
|
export * from './policy.js';
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { type TaskDifficulty } from './tasks.js';
|
|
3
|
+
export declare const MODEL_COMPARISON: {
|
|
4
|
+
readonly beforeDays: 30;
|
|
5
|
+
readonly afterDays: 30;
|
|
6
|
+
readonly minTasks: 20;
|
|
7
|
+
readonly rateDelta: 0.1;
|
|
8
|
+
readonly creditRatio: 0.15;
|
|
9
|
+
};
|
|
10
|
+
export declare const MODEL_VERDICTS: readonly ['better', 'worse', 'same', 'insufficient'];
|
|
11
|
+
export declare const ModelVerdict: z.ZodEnum<{
|
|
12
|
+
better: "better";
|
|
13
|
+
insufficient: "insufficient";
|
|
14
|
+
same: "same";
|
|
15
|
+
worse: "worse";
|
|
16
|
+
}>;
|
|
17
|
+
export type ModelVerdict = z.infer<typeof ModelVerdict>;
|
|
18
|
+
export type ModelComparisonSide = {
|
|
19
|
+
verified: number;
|
|
20
|
+
failed: number;
|
|
21
|
+
rejected: number;
|
|
22
|
+
credit: number | null;
|
|
23
|
+
};
|
|
24
|
+
export declare const completionRate: (s: ModelComparisonSide) => number;
|
|
25
|
+
export declare function verdictOf(before: ModelComparisonSide, after: ModelComparisonSide): ModelVerdict;
|
|
26
|
+
export declare const SUSPECTED_CHANGE: {
|
|
27
|
+
readonly recentDays: 7;
|
|
28
|
+
readonly beforeDays: 30;
|
|
29
|
+
readonly minTasks: 20;
|
|
30
|
+
readonly rateDelta: 0.05;
|
|
31
|
+
readonly minZ: 3;
|
|
32
|
+
readonly creditRatio: 0.15;
|
|
33
|
+
readonly lapseDays: 30;
|
|
34
|
+
};
|
|
35
|
+
export type SuspectedChangeRule = {
|
|
36
|
+
readonly minTasks: number;
|
|
37
|
+
readonly rateDelta: number;
|
|
38
|
+
readonly minZ: number;
|
|
39
|
+
readonly creditRatio: number;
|
|
40
|
+
};
|
|
41
|
+
export declare const SUSPECTED_STATES: readonly ['open', 'confirmed_declared', 'confirmed_network', 'dismissed', 'lapsed'];
|
|
42
|
+
export declare const SuspectedState: z.ZodEnum<{
|
|
43
|
+
confirmed_declared: "confirmed_declared";
|
|
44
|
+
confirmed_network: "confirmed_network";
|
|
45
|
+
dismissed: "dismissed";
|
|
46
|
+
lapsed: "lapsed";
|
|
47
|
+
open: "open";
|
|
48
|
+
}>;
|
|
49
|
+
export type SuspectedState = z.infer<typeof SuspectedState>;
|
|
50
|
+
export type SuspectedLevels = Partial<Record<TaskDifficulty, ModelComparisonSide>>;
|
|
51
|
+
export type SuspectedShift = {
|
|
52
|
+
shifted: boolean;
|
|
53
|
+
rate: number | null;
|
|
54
|
+
credit: number | null;
|
|
55
|
+
z: number | null;
|
|
56
|
+
};
|
|
57
|
+
export declare function shiftOf(before: SuspectedLevels, after: SuspectedLevels, rule?: SuspectedChangeRule): SuspectedShift;
|
|
58
|
+
export declare const MODEL_NETWORK: {
|
|
59
|
+
readonly minAgents: 3;
|
|
60
|
+
readonly minOperators: 2;
|
|
61
|
+
readonly perOperator: 3;
|
|
62
|
+
readonly models: 50;
|
|
63
|
+
readonly events: 20;
|
|
64
|
+
readonly changes: 20;
|
|
65
|
+
};
|
|
66
|
+
export type ModelVerdictCounts = {
|
|
67
|
+
better: number;
|
|
68
|
+
worse: number;
|
|
69
|
+
same: number;
|
|
70
|
+
};
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { TASK_DIFFICULTIES } from './tasks.js';
|
|
3
|
+
/*
|
|
4
|
+
* The model change comparison (VOU-551, UI-23, D-UI-6). When an agent
|
|
5
|
+
* declares another model (agent_model_changes, VOU-566), the nightly
|
|
6
|
+
* scoring run reads the agent's own tasks as claimant in the beforeDays
|
|
7
|
+
* before the change and in the days since, up to afterDays, per task
|
|
8
|
+
* category, and stores what it found beside the change. A report only.
|
|
9
|
+
* It moves no Trust Score, no category score, no stored credit and no
|
|
10
|
+
* level, and nothing reads it for standing.
|
|
11
|
+
*
|
|
12
|
+
* One side of one category holds four numbers. verified is the verified
|
|
13
|
+
* tasks that count for Trust, failed the claims ended at the failed submit
|
|
14
|
+
* cap or released after a failed submit, never a clean release (VOU-577),
|
|
15
|
+
* rejected the tasks whose poster answered the agent's success with
|
|
16
|
+
* failure, and credit the average stored base credit
|
|
17
|
+
* of the verified tasks that store one, null when none does.
|
|
18
|
+
*
|
|
19
|
+
* The rule, proposed by the VOU-551 pull request for Carl to agree before
|
|
20
|
+
* it ships, so the numbers live here and nowhere else. The completion rate
|
|
21
|
+
* is verified over verified plus failed plus rejected. Below minTasks
|
|
22
|
+
* verified tasks on either side there is no verdict, insufficient, since a
|
|
23
|
+
* rate over a handful of tasks swings on one of them. From there a side is
|
|
24
|
+
* better when the rate rose by at least rateDelta or the credit by at
|
|
25
|
+
* least creditRatio of the before side's, and neither fell by those
|
|
26
|
+
* amounts, worse the mirror, and same otherwise, so a rate that rose while
|
|
27
|
+
* the credit fell reads same. Ten points of completion is two tasks in
|
|
28
|
+
* twenty, so at minTasks one task alone never moves the rate to a
|
|
29
|
+
* verdict. Fifteen percent of credit is a starting value, to tune on real
|
|
30
|
+
* data like the Trust thresholds. Both are compared after rounding to six
|
|
31
|
+
* decimals, so 0.9 less 0.8 counts as the ten points it is.
|
|
32
|
+
*/
|
|
33
|
+
export const MODEL_COMPARISON = {
|
|
34
|
+
beforeDays: 30,
|
|
35
|
+
afterDays: 30,
|
|
36
|
+
minTasks: 20,
|
|
37
|
+
rateDelta: 0.1,
|
|
38
|
+
creditRatio: 0.15,
|
|
39
|
+
};
|
|
40
|
+
export const MODEL_VERDICTS = [
|
|
41
|
+
'better',
|
|
42
|
+
'worse',
|
|
43
|
+
'same',
|
|
44
|
+
'insufficient',
|
|
45
|
+
];
|
|
46
|
+
export const ModelVerdict = z.enum(MODEL_VERDICTS);
|
|
47
|
+
const round6 = (n) => Math.round(n * 1e6) / 1e6;
|
|
48
|
+
// verified over every attempt that ended, the rule above. A side with
|
|
49
|
+
// minTasks verified tasks always has attempts, so it never divides by 0.
|
|
50
|
+
export const completionRate = (s) => s.verified / (s.verified + s.failed + s.rejected);
|
|
51
|
+
// The verdict of one category, the rule above. Pure, so the run, a test
|
|
52
|
+
// and a later reader agree.
|
|
53
|
+
export function verdictOf(before, after) {
|
|
54
|
+
const { minTasks, rateDelta, creditRatio } = MODEL_COMPARISON;
|
|
55
|
+
if (before.verified < minTasks || after.verified < minTasks) {
|
|
56
|
+
return 'insufficient';
|
|
57
|
+
}
|
|
58
|
+
const rate = round6(completionRate(after) - completionRate(before));
|
|
59
|
+
const credit = before.credit === null || after.credit === null || before.credit <= 0
|
|
60
|
+
? 0
|
|
61
|
+
: round6(after.credit / before.credit - 1);
|
|
62
|
+
const up = rate >= rateDelta || credit >= creditRatio;
|
|
63
|
+
const down = rate <= -rateDelta || credit <= -creditRatio;
|
|
64
|
+
if (up && !down)
|
|
65
|
+
return 'better';
|
|
66
|
+
if (down && !up)
|
|
67
|
+
return 'worse';
|
|
68
|
+
return 'same';
|
|
69
|
+
}
|
|
70
|
+
/*
|
|
71
|
+
* Suspected changes (VOU-562, UI-30, D-UI-10, D-UI-11). When an agent's
|
|
72
|
+
* results in a category shift and nothing it reported explains it, the
|
|
73
|
+
* nightly scoring run keeps a suspected change of model for its operator,
|
|
74
|
+
* and for nobody else. A claim SealKeeper makes about the agent, so it is
|
|
75
|
+
* on no public route, in no feed item and in no SEAL, and it moves no
|
|
76
|
+
* Trust Score, no category score, no stored credit and no level.
|
|
77
|
+
*
|
|
78
|
+
* The windows, per UTC day of the run. after is the recentDays before the
|
|
79
|
+
* start of the day, before the beforeDays before that. Each side counts
|
|
80
|
+
* per difficulty level what the model comparison counts per side
|
|
81
|
+
* (ModelComparisonSide), the agent's own tasks as claimant.
|
|
82
|
+
*
|
|
83
|
+
* The shift, compared at the same difficulty. A level counts only with at
|
|
84
|
+
* least one attempt on each side, an attempt being a verified, failed or
|
|
85
|
+
* rejected task. Over the levels L that count, with w_d the attempts of
|
|
86
|
+
* level d on both sides and v_d its verified tasks on both sides,
|
|
87
|
+
*
|
|
88
|
+
* rate = sum over L of w_d * (rate_after_d - rate_before_d) / sum of w_d
|
|
89
|
+
* credit = sum over L' of v_d * (credit_after_d / credit_before_d - 1)
|
|
90
|
+
* / sum of v_d, L' the levels of L with a credit on both sides
|
|
91
|
+
* z = |rate| / sqrt(p * (1 - p) * (1 / n_before + 1 / n_after))
|
|
92
|
+
*
|
|
93
|
+
* rate_x_d being completionRate of level d on side x, n_x the attempts of
|
|
94
|
+
* L on side x and p the verified tasks of L on both sides over all their
|
|
95
|
+
* attempts. So a shift that comes only from harder or easier tasks reads
|
|
96
|
+
* as none, since every level is compared with itself. Below minTasks
|
|
97
|
+
* verified tasks of L on either side there is no shift, since tasks of a
|
|
98
|
+
* level with nothing to compare against say nothing about one. From there
|
|
99
|
+
* the results shifted when the rate moved by rateDelta or more with z at
|
|
100
|
+
* minZ or more, or the credit moved by creditRatio of the before side's or
|
|
101
|
+
* more, either way, compared after rounding to six decimals.
|
|
102
|
+
*
|
|
103
|
+
* The numbers wait for Carl's agreement at the VOU-562 pull request. The
|
|
104
|
+
* issue proposed 5 points of completion or 15 percent of credit alone.
|
|
105
|
+
* Over seeded steady history, one agent with the same chance per level
|
|
106
|
+
* every day, 5 points alone fired on about half the nights at 5 tasks a
|
|
107
|
+
* day and on a third at 10, since 5 points is one task in twenty. minZ is
|
|
108
|
+
* the tuning the seeded history asked for, a shift of 3 standard errors
|
|
109
|
+
* of the difference, which fired on under 0.4 percent of the nights at 3
|
|
110
|
+
* to 20 tasks a day, and on none of the 24 nights of the 60 days the API
|
|
111
|
+
* test seeds, while a real drop of 20 points is caught on most nights at
|
|
112
|
+
* 10 tasks a day or more. The credit side had no variance in the seeded
|
|
113
|
+
* history, where every task of a level earns the same.
|
|
114
|
+
*
|
|
115
|
+
* A suspected change stays open lapseDays at most. A declared change of
|
|
116
|
+
* model or a network event for the agent's model confirms it sooner, and
|
|
117
|
+
* the operator can dismiss it.
|
|
118
|
+
*/
|
|
119
|
+
export const SUSPECTED_CHANGE = {
|
|
120
|
+
recentDays: 7,
|
|
121
|
+
beforeDays: 30,
|
|
122
|
+
minTasks: 20,
|
|
123
|
+
rateDelta: 0.05,
|
|
124
|
+
minZ: 3,
|
|
125
|
+
creditRatio: 0.15,
|
|
126
|
+
lapseDays: 30,
|
|
127
|
+
};
|
|
128
|
+
// A suspected change's life. open until one of the others, each final.
|
|
129
|
+
export const SUSPECTED_STATES = [
|
|
130
|
+
'open',
|
|
131
|
+
'confirmed_declared',
|
|
132
|
+
'confirmed_network',
|
|
133
|
+
'dismissed',
|
|
134
|
+
'lapsed',
|
|
135
|
+
];
|
|
136
|
+
export const SuspectedState = z.enum(SUSPECTED_STATES);
|
|
137
|
+
// The shift of one category, the rule above. Pure, so the run, a test and
|
|
138
|
+
// a later reader agree. rule is SUSPECTED_CHANGE unless a test tunes it.
|
|
139
|
+
export function shiftOf(before, after, rule = SUSPECTED_CHANGE) {
|
|
140
|
+
const attempts = (s) => s.verified + s.failed + s.rejected;
|
|
141
|
+
let weight = 0;
|
|
142
|
+
let rateSum = 0;
|
|
143
|
+
let creditWeight = 0;
|
|
144
|
+
let creditSum = 0;
|
|
145
|
+
let verifiedBefore = 0;
|
|
146
|
+
let verifiedAfter = 0;
|
|
147
|
+
let attemptsBefore = 0;
|
|
148
|
+
let attemptsAfter = 0;
|
|
149
|
+
for (const d of TASK_DIFFICULTIES) {
|
|
150
|
+
const b = before[d];
|
|
151
|
+
const a = after[d];
|
|
152
|
+
if (!b || !a || attempts(b) === 0 || attempts(a) === 0)
|
|
153
|
+
continue;
|
|
154
|
+
const w = attempts(b) + attempts(a);
|
|
155
|
+
weight += w;
|
|
156
|
+
rateSum += w * (completionRate(a) - completionRate(b));
|
|
157
|
+
verifiedBefore += b.verified;
|
|
158
|
+
verifiedAfter += a.verified;
|
|
159
|
+
attemptsBefore += attempts(b);
|
|
160
|
+
attemptsAfter += attempts(a);
|
|
161
|
+
if (b.credit !== null && a.credit !== null && b.credit > 0) {
|
|
162
|
+
const v = b.verified + a.verified;
|
|
163
|
+
creditWeight += v;
|
|
164
|
+
creditSum += v * (a.credit / b.credit - 1);
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
if (verifiedBefore < rule.minTasks || verifiedAfter < rule.minTasks) {
|
|
168
|
+
return { shifted: false, rate: null, credit: null, z: null };
|
|
169
|
+
}
|
|
170
|
+
const rate = round6(rateSum / weight);
|
|
171
|
+
const credit = creditWeight > 0 ? round6(creditSum / creditWeight) : null;
|
|
172
|
+
const p = (verifiedBefore + verifiedAfter) / (attemptsBefore + attemptsAfter);
|
|
173
|
+
const se = Math.sqrt(p * (1 - p) * (1 / attemptsBefore + 1 / attemptsAfter));
|
|
174
|
+
const z = se > 0 ? round6(Math.abs(rate) / se) : 0;
|
|
175
|
+
const shifted = (Math.abs(rate) >= rule.rateDelta && z >= rule.minZ) ||
|
|
176
|
+
(credit !== null && Math.abs(credit) >= rule.creditRatio);
|
|
177
|
+
return { shifted, rate, credit, z };
|
|
178
|
+
}
|
|
179
|
+
/*
|
|
180
|
+
* Network events (VOU-551). Where minAgents or more agents, of at least
|
|
181
|
+
* minOperators operators, recorded the same change, the same old modelKey
|
|
182
|
+
* to the same new one, in the same UTC week from Monday (trustWeekOf), the
|
|
183
|
+
* nightly run keeps one model_network_events row with the count of those
|
|
184
|
+
* agents' verdicts per category. An agent with no declared name on either
|
|
185
|
+
* side is in none, and neither is a change of the fingerprint's model part
|
|
186
|
+
* alone, whose two keys are the same. Two operators, so one operator's own
|
|
187
|
+
* agents never make an event alone.
|
|
188
|
+
*
|
|
189
|
+
* perOperator (VOU-578). An event counts at most this many agents of any
|
|
190
|
+
* one operator, in agents and in its verdicts, the ones with the lowest
|
|
191
|
+
* agent ids, the rest left out. An operator may run 100 agents, so without
|
|
192
|
+
* it one operator with 99 agents and one colluding agent would make an
|
|
193
|
+
* event and supply almost every verdict. 3 is minAgents, so one
|
|
194
|
+
* operator's agents never count for more than the size an event needs,
|
|
195
|
+
* and in an event of 2 operators the other's agents are at least a
|
|
196
|
+
* quarter of the verdicts.
|
|
197
|
+
*
|
|
198
|
+
* The caps of the three reads, GET /v1/models (models and events) and GET
|
|
199
|
+
* /v1/agents/:id/model-changes (changes).
|
|
200
|
+
*/
|
|
201
|
+
export const MODEL_NETWORK = {
|
|
202
|
+
minAgents: 3,
|
|
203
|
+
minOperators: 2,
|
|
204
|
+
perOperator: 3,
|
|
205
|
+
models: 50,
|
|
206
|
+
events: 20,
|
|
207
|
+
changes: 20,
|
|
208
|
+
};
|