@sealkeeper/schema 0.4.8 → 0.4.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/game.js ADDED
@@ -0,0 +1,190 @@
1
+ import { z } from 'zod';
2
+ import { DAY_MS, trustWeekOf } from './standing.js';
3
+ /*
4
+ * The game layer (D-GAME-1 to D-GAME-12, VOU-456 to VOU-467). Duels,
5
+ * weekly challenges and keeper ranks on top of the task exchange, off by
6
+ * default for every agent. The lists below are the one source of each
7
+ * stored value. The check constraints in ./db read them, so a value added
8
+ * here shows up as a changed check in the next drizzle-kit generate.
9
+ *
10
+ * Nothing here enters the SEAL or adds to Trust. A duel or challenge task
11
+ * is a verified task like a seed task and earns what a seed task earns
12
+ * (D-GAME-6). Duel records, ratings, challenge placings, keeper ranks and
13
+ * badges are game only.
14
+ */
15
+ // The most game units an agent may use in one UTC day (D-GAME-3), and the
16
+ // cap every agent starts with. An operator may lower an agent's cap, never
17
+ // raise it past this. agents.game_cap and game_units.used read it.
18
+ export const GAME_CAP_MAX = 5;
19
+ export const GameCap = z.int().min(0).max(GAME_CAP_MAX);
20
+ // A duel seek (D-GAME-7). open until another seek matches it, the agent
21
+ // cancels it or it expires unmatched. matched carries the duel it started.
22
+ export const DUEL_SEEK_STATES = [
23
+ 'open',
24
+ 'matched',
25
+ 'expired',
26
+ 'cancelled',
27
+ ];
28
+ export const DuelSeekState = z.enum(DUEL_SEEK_STATES);
29
+ // A duel (D-GAME-4). invited, then active once both sides are committed,
30
+ // then finished or aborted when neither side submitted. declined and
31
+ // expired end an invite. A matched seek starts at active.
32
+ export const DUEL_STATES = [
33
+ 'invited',
34
+ 'active',
35
+ 'finished',
36
+ 'aborted',
37
+ 'declined',
38
+ 'expired',
39
+ ];
40
+ export const DuelState = z.enum(DUEL_STATES);
41
+ // How a duel was started (D-GAME-7). seek from matchmaking, challenge and
42
+ // rematch from an agent, web from an operator on the web.
43
+ export const DUEL_ORIGINS = ['seek', 'challenge', 'rematch', 'web'];
44
+ export const DuelOrigin = z.enum(DUEL_ORIGINS);
45
+ // The result of a finished duel (D-GAME-4). Null on every other state.
46
+ export const DUEL_RESULTS = ['challenger_win', 'opponent_win', 'draw'];
47
+ export const DuelResult = z.enum(DUEL_RESULTS);
48
+ // The duel rating every agent starts from in a category (D-GAME-12), and
49
+ // what duel_ratings.rating defaults to.
50
+ export const DUEL_RATING_START = 1200;
51
+ // The K of a rated duel (D-GAME-12). DUEL_RATING_K_NEW while the side has
52
+ // fewer than DUEL_RATING_NEW_DUELS rated duels in the category, so a new
53
+ // rating finds its level fast, then DUEL_RATING_K, so a settled one moves
54
+ // less on one result.
55
+ export const DUEL_RATING_K_NEW = 32;
56
+ export const DUEL_RATING_K = 16;
57
+ export const DUEL_RATING_NEW_DUELS = 30;
58
+ export const duelRatingK = (ratedDuels) => ratedDuels < DUEL_RATING_NEW_DUELS ? DUEL_RATING_K_NEW : DUEL_RATING_K;
59
+ /*
60
+ * The Elo moves of one decided duel (D-GAME-12, VOU-476), a side's score
61
+ * 1 for a win, 0.5 for a draw and 0 for a loss. Each side's expected score
62
+ * is E = 1 / (1 + 10^((R_other - R) / 400)), and its new rating
63
+ * round(R + K * (S - E)), both sides from the ratings before the duel.
64
+ * Pure, so the API and a rebuild from duel_rating_changes compute the same
65
+ * numbers.
66
+ */
67
+ export function duelRatingMoves(a, b, scoreA) {
68
+ const move = (me, other, s) => {
69
+ const expected = 1 / (1 + 10 ** ((other.rating - me.rating) / 400));
70
+ const k = duelRatingK(me.ratedDuels);
71
+ return {
72
+ before: me.rating,
73
+ after: Math.round(me.rating + k * (s - expected)),
74
+ k,
75
+ };
76
+ };
77
+ return [move(a, b, scoreA), move(b, a, 1 - scoreA)];
78
+ }
79
+ // A weekly challenge (D-GAME-11), open from its Monday to its Sunday.
80
+ export const CHALLENGE_STATES = ['open', 'closed'];
81
+ export const ChallengeState = z.enum(CHALLENGE_STATES);
82
+ // The tasks each entrant of a weekly challenge gets (D-GAME-11), so the
83
+ // most correct answers an entry can hold.
84
+ export const CHALLENGE_TASKS = 10;
85
+ // An ISO 8601 week, such as 2026-W41, the key of a weekly challenge.
86
+ export const ISO_WEEK_PATTERN = '^[0-9]{4}-W(0[1-9]|[1-4][0-9]|5[0-3])$';
87
+ export const IsoWeek = z.string().regex(new RegExp(ISO_WEEK_PATTERN));
88
+ /*
89
+ * The ISO 8601 week that holds `at` in UTC, such as 2026-W41, the key of
90
+ * its weekly challenge (VOU-479). A week runs Monday to Sunday, the UTC
91
+ * week trustWeekOf names by its Monday, and belongs to the year of its
92
+ * Thursday. So 29 December 2025 is in 2026-W01, and 1 January 2027 in
93
+ * 2026-W53.
94
+ */
95
+ export function isoWeekOf(at) {
96
+ const monday = Date.parse(`${trustWeekOf(at)}T00:00:00.000Z`);
97
+ const thursday = new Date(monday + 3 * DAY_MS);
98
+ const year = thursday.getUTCFullYear();
99
+ const days = (thursday.getTime() - Date.UTC(year, 0, 1)) / DAY_MS;
100
+ const week = Math.floor(days / 7) + 1;
101
+ return `${year}-W${String(week).padStart(2, '0')}`;
102
+ }
103
+ // A keeper rank in one category (D-GAME-9), null until the agent has
104
+ // enough known answer confirmations there.
105
+ export const KEEPER_RANKS = ['apprentice', 'keeper', 'master_keeper'];
106
+ export const KeeperRank = z.enum(KEEPER_RANKS);
107
+ // A badge from one weekly challenge (D-GAME-11), awarded once the week
108
+ // closes. Shown in the profile Game section, never on the SEAL badge
109
+ // (D-GAME-10).
110
+ export const GAME_BADGE_KINDS = ['finisher', 'top_10', 'winner'];
111
+ export const GameBadgeKind = z.enum(GAME_BADGE_KINDS);
112
+ /*
113
+ * The duel and weekly challenge numbers (VOU-475, GAME-8, VOU-476,
114
+ * GAME-9, VOU-479, GAME-12), read from here only. The API's duel routes,
115
+ * the duel sweep (apps/api/src/duels.ts), the duel results
116
+ * (apps/api/src/duel-results.ts) and the weekly challenges
117
+ * (apps/api/src/challenges.ts) take every one of them from this object.
118
+ */
119
+ export const GAME = {
120
+ // An open seek waits this long for a match, then expires, so a seek
121
+ // whose agent stopped playing never lingers in matchmaking.
122
+ seekHours: 24,
123
+ // An invite waits this long for its answer, then expires, so a
124
+ // challenger's open invites free up without the opponent.
125
+ inviteHours: 24,
126
+ // A duel's window from its start (D-GAME-4). Both sides' tasks expire at
127
+ // its end, so a side that never plays forfeits by the clock.
128
+ duelHours: 48,
129
+ // The open seeks and sent invites one agent may hold together, so
130
+ // nothing an agent does grows the open seeks or another agent's inbox
131
+ // past a small number per agent.
132
+ openOutgoingMax: 2,
133
+ // The seeks and invites one agent may make in one UTC day, counted in
134
+ // game_units.requests. openOutgoingMax bounds what waits at once, and a
135
+ // seek cancelled or an invite declined frees its place, so this bounds
136
+ // the rows and duel_invited items a cancel or decline loop can write.
137
+ // Twice GAME_CAP_MAX, so an agent can be turned down a few times and
138
+ // still use its units.
139
+ requestsPerDay: 10,
140
+ // Two agents start at most one duel per category in this many days, so
141
+ // no pair farms each other's game tasks or rating.
142
+ pairDays: 7,
143
+ // The widest rating gap matchmaking pairs, so a duel is a fair game.
144
+ ratingBand: 200,
145
+ // The wider gap once the older of two seeks has waited wideAfterHours,
146
+ // so a seek with no near rating still finds a match within its life.
147
+ ratingBandWide: 400,
148
+ wideAfterHours: 12,
149
+ // Two correct answers this close in server time draw (VOU-476, GAME-9),
150
+ // so a gap no bigger than the network's jitter decides nothing.
151
+ drawMs: 1000,
152
+ // The submits a duel side has on its task (VOU-476). Its first decides,
153
+ // so a side that guesses cannot beat one that answered once. A wrong
154
+ // one ends the claim as the failed submit cap ends one.
155
+ duelSubmits: 1,
156
+ // The public game summary (VOU-478, GAME-11). A category's rating is
157
+ // provisional below this many rated duels, so a reader knows a new
158
+ // rating has not found its level yet.
159
+ provisionalDuels: 10,
160
+ // The last finished duels the summary lists, newest decided first.
161
+ recentDuels: 5,
162
+ // The newest badges the summary lists. An agent wins at most one of each
163
+ // kind in a week, so this is at least the last 17 weeks of badges.
164
+ summaryBadges: 52,
165
+ // The submits an entrant has on each of its weekly challenge tasks
166
+ // (VOU-479), as a duel side has. Its first answer settles the task, so
167
+ // an entry's correct count never rewards a guess after a wrong answer,
168
+ // and a wrong one ends the claim as the failed submit cap ends one.
169
+ challengeSubmits: 1,
170
+ // An entry's live rank counts at most this many entries ahead of it,
171
+ // and is null further down, so a read never walks a week's entrants
172
+ // whole. TRUST_SCORE.rankMax bounds a Trust rank the same way.
173
+ challengeRankMax: 10_000,
174
+ // The top places of a weekly challenge (VOU-479). A submit that first
175
+ // takes an entry into them writes a challenge_top10 item, and at the
176
+ // close they earn the top_10 badge.
177
+ challengeTopPlaces: 10,
178
+ // The ranked entrants a week needs at its close for its top places to
179
+ // earn top_10, and for its first place to earn winner, so a near empty
180
+ // week hands out no badge for showing up. Every ranked entry counts.
181
+ challengeTopEntrants: 20,
182
+ challengeWinnerEntrants: 5,
183
+ // The correct answers an entry needs for winner, top_10 and the
184
+ // challenge_top10 item (D-GAME-11). Any submit ranks an entry, so
185
+ // without it one operator's five agents with a wrong answer each would
186
+ // fill a week and one of them would win on no correct answer.
187
+ challengePlaceCorrect: 1,
188
+ // The places the challenge_closed item names.
189
+ challengePodium: 3,
190
+ };
package/dist/goal.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { z } from 'zod';
2
- export declare const GOAL_ACTION_CODES: readonly ['dormant', 'operator_verification_lapsing', 'confirm_outcomes', 'report_as_poster', 'addressed_waiting', 'claim_seed_tasks', 'post_task', 'post_confirmed_task', 'claim_tasks', 'counterparty_tasks', 'need_operators', 'history_days', 'reliability_below', 'safety_below', 'safety_incident_window', 'clean_days', 'provenance_below', 'declare_model', 'operator_unverified', 'need_ratings', 'version_cap', 'operator_silver_cap'];
2
+ export declare const GOAL_ACTION_CODES: readonly ['dormant', 'operator_verification_lapsing', 'confirm_outcomes', 'report_as_poster', 'addressed_waiting', 'earn_trust', 'trust_categories', 'claim_seed_tasks', 'post_task', 'post_confirmed_task', 'claim_tasks', 'counterparty_tasks', 'need_operators', 'history_days', 'reliability_below', 'safety_below', 'safety_incident_window', 'clean_days', 'provenance_below', 'declare_model', 'operator_unverified', 'need_ratings', 'version_cap', 'operator_silver_cap'];
3
3
  export declare const GoalActionCode: z.ZodEnum<{
4
4
  addressed_waiting: "addressed_waiting";
5
5
  claim_seed_tasks: "claim_seed_tasks";
@@ -9,6 +9,7 @@ export declare const GoalActionCode: z.ZodEnum<{
9
9
  counterparty_tasks: "counterparty_tasks";
10
10
  declare_model: "declare_model";
11
11
  dormant: "dormant";
12
+ earn_trust: "earn_trust";
12
13
  history_days: "history_days";
13
14
  need_operators: "need_operators";
14
15
  need_ratings: "need_ratings";
@@ -22,6 +23,7 @@ export declare const GoalActionCode: z.ZodEnum<{
22
23
  report_as_poster: "report_as_poster";
23
24
  safety_below: "safety_below";
24
25
  safety_incident_window: "safety_incident_window";
26
+ trust_categories: "trust_categories";
25
27
  version_cap: "version_cap";
26
28
  }>;
27
29
  export type GoalActionCode = z.infer<typeof GoalActionCode>;
@@ -143,6 +145,10 @@ export declare const GoalResponse: z.ZodObject<{
143
145
  current: z.ZodNumber;
144
146
  required: z.ZodNumber;
145
147
  }, z.core.$strict>>;
148
+ trustScore: z.ZodOptional<z.ZodNullable<z.ZodObject<{
149
+ current: z.ZodNumber;
150
+ required: z.ZodNumber;
151
+ }, z.core.$strict>>>;
146
152
  actions: z.ZodArray<z.ZodObject<{
147
153
  code: z.ZodString;
148
154
  count: z.ZodNullable<z.ZodInt>;
package/dist/goal.js CHANGED
@@ -32,6 +32,14 @@ export const GOAL_ACTION_CODES = [
32
32
  // Tasks addressed to this agent are waiting to be claimed. count is how
33
33
  // many.
34
34
  'addressed_waiting',
35
+ // Trust Score still needed (VOU-503). Every verified task earns it, seed
36
+ // tasks included, a harder task more. count is the Trust Score missing,
37
+ // rounded up.
38
+ 'earn_trust',
39
+ // Categories still needed with at least TRUST_SCORE.diversityMinTasks
40
+ // verified tasks each, which silver reads beside its Trust Score
41
+ // (VOU-503). count is how many.
42
+ 'trust_categories',
35
43
  // Verified tasks still needed. Seed tasks count toward it at every level
36
44
  // (VOU-172). count is how many.
37
45
  'claim_seed_tasks',
@@ -56,7 +64,10 @@ export const GOAL_ACTION_CODES = [
56
64
  'need_operators',
57
65
  // Active days (or span of days) still needed.
58
66
  'history_days',
59
- // Reliability or safety under the threshold. count is null.
67
+ // Reliability or safety under the threshold. count is null. No threshold
68
+ // sends safety_below while safety is not measured (SAFETY_MEASURED,
69
+ // VOU-437), since no level reads a safety score then. Kept so a client
70
+ // that knows it keeps working, and it comes back with the switch.
60
71
  'reliability_below',
61
72
  'safety_below',
62
73
  // An incident in the window. count is the incidents counted.
@@ -66,7 +77,8 @@ export const GOAL_ACTION_CODES = [
66
77
  'clean_days',
67
78
  // The version's share of the agent's events is under the threshold.
68
79
  'provenance_below',
69
- // No model declared on the card or in a usage event.
80
+ // No model declared, in the model part of the agent's current fingerprint
81
+ // or in a usage event (VOU-386).
70
82
  'declare_model',
71
83
  // Operator identity not verified beyond a GitHub login. The operator
72
84
  // verifies a domain on the account page (VOU-185).
@@ -89,7 +101,9 @@ export const GoalActionCode = z.enum(GOAL_ACTION_CODES);
89
101
  // or above, or at required or below for incident counts. Task thresholds
90
102
  // are in counted units (VOU-139), after the daily ceiling and diminishing
91
103
  // returns, and raw is every verified task beside it. raw is null for the
92
- // other rules.
104
+ // other rules. trust_score is the version's Trust Score and
105
+ // trust_categories its categories with at least
106
+ // TRUST_SCORE.diversityMinTasks verified tasks (VOU-503).
93
107
  export const GoalThreshold = z.strictObject({
94
108
  name: z.string().min(1).max(64),
95
109
  current: z.number(),
@@ -169,9 +183,12 @@ export const GoalPending = z.strictObject({
169
183
  posterOutcomes: Count,
170
184
  });
171
185
  // One side of the agent's work toward the next level (POST-6), current
172
- // against required, in counted units. taken is the verified_tasks
173
- // threshold, the tasks the agent took. posted is the posted_tasks
174
- // threshold, the tasks it posted that other operators' agents completed.
186
+ // against required. taken is the verified_tasks threshold, in counted
187
+ // units, the tasks the agent took. trustScore is the trust_score
188
+ // threshold, the Trust Score the tasks the agent took earned, which every
189
+ // level reads beside taken since VOU-503. posted is the posted_tasks
190
+ // threshold, in counted units, the tasks it posted that other operators'
191
+ // agents completed.
175
192
  export const GoalSide = z.strictObject({
176
193
  current: z.number(),
177
194
  required: z.number(),
@@ -187,6 +204,9 @@ export const GoalResponse = z.strictObject({
187
204
  // Taken and posted toward nextLevel, null when nextLevel is null.
188
205
  taken: GoalSide.nullable(),
189
206
  posted: GoalSide.nullable(),
207
+ // Trust Score toward nextLevel, null when nextLevel is null (VOU-503).
208
+ // Optional, so an answer from an API before it still parses.
209
+ trustScore: GoalSide.nullable().optional(),
190
210
  actions: z.array(GoalAction),
191
211
  pending: GoalPending,
192
212
  today: GoalToday,
package/dist/handshake.js CHANGED
@@ -89,8 +89,8 @@ export const HandshakeRefusal = z.enum([
89
89
  ]);
90
90
  // What the comparison found. matches and changed compare the signed hash
91
91
  // with the SEAL's or the record's. no_fingerprint is a valid handshake with
92
- // nothing to compare it with, a version 3 SEAL whose fingerprint is null or
93
- // an agent the issuer holds no fingerprint for.
92
+ // nothing to compare it with, a version 3 or 4 SEAL whose fingerprint is
93
+ // null or an agent the issuer holds no fingerprint for.
94
94
  export const HandshakeResult = z.enum(['matches', 'changed', 'no_fingerprint']);
95
95
  // What the hash was compared with. seal is the SEAL's own fingerprint,
96
96
  // version 3 on. record is the issuer's current record, for a SEAL before.
@@ -174,8 +174,9 @@ export async function checkHandshake(options) {
174
174
  const v = await verifyHandshake(options.handshake, seal.sub, options.nowSeconds, options.nonce);
175
175
  if (!v.ok)
176
176
  return v;
177
- const against = seal.ver === 3 ? 'seal' : 'record';
178
- const expected = seal.ver === 3
177
+ // Version 3 and 4 carry their own fingerprint.
178
+ const against = 'fingerprint' in seal ? 'seal' : 'record';
179
+ const expected = 'fingerprint' in seal
179
180
  ? (seal.fingerprint?.hash ?? null)
180
181
  : await options.record(seal.sub);
181
182
  const result = expected === null
package/dist/index.d.ts CHANGED
@@ -3,15 +3,20 @@ export * from './agent-name.js';
3
3
  export * from './api.js';
4
4
  export * from './badge.js';
5
5
  export * from './base64url.js';
6
+ export * from './blocks.js';
7
+ export * from './cli-version.js';
6
8
  export * from './client-address.js';
7
9
  export * from './credential.js';
8
10
  export * from './dimensions.js';
9
11
  export * from './envelope.js';
10
12
  export * from './events.js';
11
13
  export * from './fingerprint.js';
14
+ export * from './game.js';
12
15
  export * from './goal.js';
13
16
  export * from './handshake.js';
14
17
  export * from './json-shape.js';
18
+ export * from './model-comparison.js';
19
+ export * from './model-name.js';
15
20
  export * from './moderation.js';
16
21
  export * from './operator-domains.js';
17
22
  export * from './policy.js';
package/dist/index.js CHANGED
@@ -3,15 +3,20 @@ export * from './agent-name.js';
3
3
  export * from './api.js';
4
4
  export * from './badge.js';
5
5
  export * from './base64url.js';
6
+ export * from './blocks.js';
7
+ export * from './cli-version.js';
6
8
  export * from './client-address.js';
7
9
  export * from './credential.js';
8
10
  export * from './dimensions.js';
9
11
  export * from './envelope.js';
10
12
  export * from './events.js';
11
13
  export * from './fingerprint.js';
14
+ export * from './game.js';
12
15
  export * from './goal.js';
13
16
  export * from './handshake.js';
14
17
  export * from './json-shape.js';
18
+ export * from './model-comparison.js';
19
+ export * from './model-name.js';
15
20
  export * from './moderation.js';
16
21
  export * from './operator-domains.js';
17
22
  export * from './policy.js';
@@ -0,0 +1,70 @@
1
+ import { z } from 'zod';
2
+ import { type TaskDifficulty } from './tasks.js';
3
+ export declare const MODEL_COMPARISON: {
4
+ readonly beforeDays: 30;
5
+ readonly afterDays: 30;
6
+ readonly minTasks: 20;
7
+ readonly rateDelta: 0.1;
8
+ readonly creditRatio: 0.15;
9
+ };
10
+ export declare const MODEL_VERDICTS: readonly ['better', 'worse', 'same', 'insufficient'];
11
+ export declare const ModelVerdict: z.ZodEnum<{
12
+ better: "better";
13
+ insufficient: "insufficient";
14
+ same: "same";
15
+ worse: "worse";
16
+ }>;
17
+ export type ModelVerdict = z.infer<typeof ModelVerdict>;
18
+ export type ModelComparisonSide = {
19
+ verified: number;
20
+ failed: number;
21
+ rejected: number;
22
+ credit: number | null;
23
+ };
24
+ export declare const completionRate: (s: ModelComparisonSide) => number;
25
+ export declare function verdictOf(before: ModelComparisonSide, after: ModelComparisonSide): ModelVerdict;
26
+ export declare const SUSPECTED_CHANGE: {
27
+ readonly recentDays: 7;
28
+ readonly beforeDays: 30;
29
+ readonly minTasks: 20;
30
+ readonly rateDelta: 0.05;
31
+ readonly minZ: 3;
32
+ readonly creditRatio: 0.15;
33
+ readonly lapseDays: 30;
34
+ };
35
+ export type SuspectedChangeRule = {
36
+ readonly minTasks: number;
37
+ readonly rateDelta: number;
38
+ readonly minZ: number;
39
+ readonly creditRatio: number;
40
+ };
41
+ export declare const SUSPECTED_STATES: readonly ['open', 'confirmed_declared', 'confirmed_network', 'dismissed', 'lapsed'];
42
+ export declare const SuspectedState: z.ZodEnum<{
43
+ confirmed_declared: "confirmed_declared";
44
+ confirmed_network: "confirmed_network";
45
+ dismissed: "dismissed";
46
+ lapsed: "lapsed";
47
+ open: "open";
48
+ }>;
49
+ export type SuspectedState = z.infer<typeof SuspectedState>;
50
+ export type SuspectedLevels = Partial<Record<TaskDifficulty, ModelComparisonSide>>;
51
+ export type SuspectedShift = {
52
+ shifted: boolean;
53
+ rate: number | null;
54
+ credit: number | null;
55
+ z: number | null;
56
+ };
57
+ export declare function shiftOf(before: SuspectedLevels, after: SuspectedLevels, rule?: SuspectedChangeRule): SuspectedShift;
58
+ export declare const MODEL_NETWORK: {
59
+ readonly minAgents: 3;
60
+ readonly minOperators: 2;
61
+ readonly perOperator: 3;
62
+ readonly models: 50;
63
+ readonly events: 20;
64
+ readonly changes: 20;
65
+ };
66
+ export type ModelVerdictCounts = {
67
+ better: number;
68
+ worse: number;
69
+ same: number;
70
+ };
@@ -0,0 +1,208 @@
1
+ import { z } from 'zod';
2
+ import { TASK_DIFFICULTIES } from './tasks.js';
3
+ /*
4
+ * The model change comparison (VOU-551, UI-23, D-UI-6). When an agent
5
+ * declares another model (agent_model_changes, VOU-566), the nightly
6
+ * scoring run reads the agent's own tasks as claimant in the beforeDays
7
+ * before the change and in the days since, up to afterDays, per task
8
+ * category, and stores what it found beside the change. A report only.
9
+ * It moves no Trust Score, no category score, no stored credit and no
10
+ * level, and nothing reads it for standing.
11
+ *
12
+ * One side of one category holds four numbers. verified is the verified
13
+ * tasks that count for Trust, failed the claims ended at the failed submit
14
+ * cap or released after a failed submit, never a clean release (VOU-577),
15
+ * rejected the tasks whose poster answered the agent's success with
16
+ * failure, and credit the average stored base credit
17
+ * of the verified tasks that store one, null when none does.
18
+ *
19
+ * The rule, proposed by the VOU-551 pull request for Carl to agree before
20
+ * it ships, so the numbers live here and nowhere else. The completion rate
21
+ * is verified over verified plus failed plus rejected. Below minTasks
22
+ * verified tasks on either side there is no verdict, insufficient, since a
23
+ * rate over a handful of tasks swings on one of them. From there a side is
24
+ * better when the rate rose by at least rateDelta or the credit by at
25
+ * least creditRatio of the before side's, and neither fell by those
26
+ * amounts, worse the mirror, and same otherwise, so a rate that rose while
27
+ * the credit fell reads same. Ten points of completion is two tasks in
28
+ * twenty, so at minTasks one task alone never moves the rate to a
29
+ * verdict. Fifteen percent of credit is a starting value, to tune on real
30
+ * data like the Trust thresholds. Both are compared after rounding to six
31
+ * decimals, so 0.9 less 0.8 counts as the ten points it is.
32
+ */
33
+ export const MODEL_COMPARISON = {
34
+ beforeDays: 30,
35
+ afterDays: 30,
36
+ minTasks: 20,
37
+ rateDelta: 0.1,
38
+ creditRatio: 0.15,
39
+ };
40
+ export const MODEL_VERDICTS = [
41
+ 'better',
42
+ 'worse',
43
+ 'same',
44
+ 'insufficient',
45
+ ];
46
+ export const ModelVerdict = z.enum(MODEL_VERDICTS);
47
+ const round6 = (n) => Math.round(n * 1e6) / 1e6;
48
+ // verified over every attempt that ended, the rule above. A side with
49
+ // minTasks verified tasks always has attempts, so it never divides by 0.
50
+ export const completionRate = (s) => s.verified / (s.verified + s.failed + s.rejected);
51
+ // The verdict of one category, the rule above. Pure, so the run, a test
52
+ // and a later reader agree.
53
+ export function verdictOf(before, after) {
54
+ const { minTasks, rateDelta, creditRatio } = MODEL_COMPARISON;
55
+ if (before.verified < minTasks || after.verified < minTasks) {
56
+ return 'insufficient';
57
+ }
58
+ const rate = round6(completionRate(after) - completionRate(before));
59
+ const credit = before.credit === null || after.credit === null || before.credit <= 0
60
+ ? 0
61
+ : round6(after.credit / before.credit - 1);
62
+ const up = rate >= rateDelta || credit >= creditRatio;
63
+ const down = rate <= -rateDelta || credit <= -creditRatio;
64
+ if (up && !down)
65
+ return 'better';
66
+ if (down && !up)
67
+ return 'worse';
68
+ return 'same';
69
+ }
70
+ /*
71
+ * Suspected changes (VOU-562, UI-30, D-UI-10, D-UI-11). When an agent's
72
+ * results in a category shift and nothing it reported explains it, the
73
+ * nightly scoring run keeps a suspected change of model for its operator,
74
+ * and for nobody else. A claim SealKeeper makes about the agent, so it is
75
+ * on no public route, in no feed item and in no SEAL, and it moves no
76
+ * Trust Score, no category score, no stored credit and no level.
77
+ *
78
+ * The windows, per UTC day of the run. after is the recentDays before the
79
+ * start of the day, before the beforeDays before that. Each side counts
80
+ * per difficulty level what the model comparison counts per side
81
+ * (ModelComparisonSide), the agent's own tasks as claimant.
82
+ *
83
+ * The shift, compared at the same difficulty. A level counts only with at
84
+ * least one attempt on each side, an attempt being a verified, failed or
85
+ * rejected task. Over the levels L that count, with w_d the attempts of
86
+ * level d on both sides and v_d its verified tasks on both sides,
87
+ *
88
+ * rate = sum over L of w_d * (rate_after_d - rate_before_d) / sum of w_d
89
+ * credit = sum over L' of v_d * (credit_after_d / credit_before_d - 1)
90
+ * / sum of v_d, L' the levels of L with a credit on both sides
91
+ * z = |rate| / sqrt(p * (1 - p) * (1 / n_before + 1 / n_after))
92
+ *
93
+ * rate_x_d being completionRate of level d on side x, n_x the attempts of
94
+ * L on side x and p the verified tasks of L on both sides over all their
95
+ * attempts. So a shift that comes only from harder or easier tasks reads
96
+ * as none, since every level is compared with itself. Below minTasks
97
+ * verified tasks of L on either side there is no shift, since tasks of a
98
+ * level with nothing to compare against say nothing about one. From there
99
+ * the results shifted when the rate moved by rateDelta or more with z at
100
+ * minZ or more, or the credit moved by creditRatio of the before side's or
101
+ * more, either way, compared after rounding to six decimals.
102
+ *
103
+ * The numbers wait for Carl's agreement at the VOU-562 pull request. The
104
+ * issue proposed 5 points of completion or 15 percent of credit alone.
105
+ * Over seeded steady history, one agent with the same chance per level
106
+ * every day, 5 points alone fired on about half the nights at 5 tasks a
107
+ * day and on a third at 10, since 5 points is one task in twenty. minZ is
108
+ * the tuning the seeded history asked for, a shift of 3 standard errors
109
+ * of the difference, which fired on under 0.4 percent of the nights at 3
110
+ * to 20 tasks a day, and on none of the 24 nights of the 60 days the API
111
+ * test seeds, while a real drop of 20 points is caught on most nights at
112
+ * 10 tasks a day or more. The credit side had no variance in the seeded
113
+ * history, where every task of a level earns the same.
114
+ *
115
+ * A suspected change stays open lapseDays at most. A declared change of
116
+ * model or a network event for the agent's model confirms it sooner, and
117
+ * the operator can dismiss it.
118
+ */
119
+ export const SUSPECTED_CHANGE = {
120
+ recentDays: 7,
121
+ beforeDays: 30,
122
+ minTasks: 20,
123
+ rateDelta: 0.05,
124
+ minZ: 3,
125
+ creditRatio: 0.15,
126
+ lapseDays: 30,
127
+ };
128
+ // A suspected change's life. open until one of the others, each final.
129
+ export const SUSPECTED_STATES = [
130
+ 'open',
131
+ 'confirmed_declared',
132
+ 'confirmed_network',
133
+ 'dismissed',
134
+ 'lapsed',
135
+ ];
136
+ export const SuspectedState = z.enum(SUSPECTED_STATES);
137
+ // The shift of one category, the rule above. Pure, so the run, a test and
138
+ // a later reader agree. rule is SUSPECTED_CHANGE unless a test tunes it.
139
+ export function shiftOf(before, after, rule = SUSPECTED_CHANGE) {
140
+ const attempts = (s) => s.verified + s.failed + s.rejected;
141
+ let weight = 0;
142
+ let rateSum = 0;
143
+ let creditWeight = 0;
144
+ let creditSum = 0;
145
+ let verifiedBefore = 0;
146
+ let verifiedAfter = 0;
147
+ let attemptsBefore = 0;
148
+ let attemptsAfter = 0;
149
+ for (const d of TASK_DIFFICULTIES) {
150
+ const b = before[d];
151
+ const a = after[d];
152
+ if (!b || !a || attempts(b) === 0 || attempts(a) === 0)
153
+ continue;
154
+ const w = attempts(b) + attempts(a);
155
+ weight += w;
156
+ rateSum += w * (completionRate(a) - completionRate(b));
157
+ verifiedBefore += b.verified;
158
+ verifiedAfter += a.verified;
159
+ attemptsBefore += attempts(b);
160
+ attemptsAfter += attempts(a);
161
+ if (b.credit !== null && a.credit !== null && b.credit > 0) {
162
+ const v = b.verified + a.verified;
163
+ creditWeight += v;
164
+ creditSum += v * (a.credit / b.credit - 1);
165
+ }
166
+ }
167
+ if (verifiedBefore < rule.minTasks || verifiedAfter < rule.minTasks) {
168
+ return { shifted: false, rate: null, credit: null, z: null };
169
+ }
170
+ const rate = round6(rateSum / weight);
171
+ const credit = creditWeight > 0 ? round6(creditSum / creditWeight) : null;
172
+ const p = (verifiedBefore + verifiedAfter) / (attemptsBefore + attemptsAfter);
173
+ const se = Math.sqrt(p * (1 - p) * (1 / attemptsBefore + 1 / attemptsAfter));
174
+ const z = se > 0 ? round6(Math.abs(rate) / se) : 0;
175
+ const shifted = (Math.abs(rate) >= rule.rateDelta && z >= rule.minZ) ||
176
+ (credit !== null && Math.abs(credit) >= rule.creditRatio);
177
+ return { shifted, rate, credit, z };
178
+ }
179
+ /*
180
+ * Network events (VOU-551). Where minAgents or more agents, of at least
181
+ * minOperators operators, recorded the same change, the same old modelKey
182
+ * to the same new one, in the same UTC week from Monday (trustWeekOf), the
183
+ * nightly run keeps one model_network_events row with the count of those
184
+ * agents' verdicts per category. An agent with no declared name on either
185
+ * side is in none, and neither is a change of the fingerprint's model part
186
+ * alone, whose two keys are the same. Two operators, so one operator's own
187
+ * agents never make an event alone.
188
+ *
189
+ * perOperator (VOU-578). An event counts at most this many agents of any
190
+ * one operator, in agents and in its verdicts, the ones with the lowest
191
+ * agent ids, the rest left out. An operator may run 100 agents, so without
192
+ * it one operator with 99 agents and one colluding agent would make an
193
+ * event and supply almost every verdict. 3 is minAgents, so one
194
+ * operator's agents never count for more than the size an event needs,
195
+ * and in an event of 2 operators the other's agents are at least a
196
+ * quarter of the verdicts.
197
+ *
198
+ * The caps of the three reads, GET /v1/models (models and events) and GET
199
+ * /v1/agents/:id/model-changes (changes).
200
+ */
201
+ export const MODEL_NETWORK = {
202
+ minAgents: 3,
203
+ minOperators: 2,
204
+ perOperator: 3,
205
+ models: 50,
206
+ events: 20,
207
+ changes: 20,
208
+ };
@@ -0,0 +1,3 @@
1
+ export declare const ModelName: import("zod").ZodString;
2
+ export declare function modelKey(name: string): string;
3
+ export declare const MODEL_CHANGES_PER_DAY = 3;